{"url":"/task/point-tracking","name":"Point Tracking","slug":"point-tracking","description_markdown":"Point Tracking, often referred to as Tracking any Point (TAP) involves acquiring, focusing on, and continuously tracking specific target point/points across video frames. The system identifies the target point, maintains focus, and predicts its movement, enabling smooth tracking even if the target moves unpredictably, or through occlusions. TAP has wide applications like object tracking, surveillance, and autonomous navigation.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":151,"papers_with_code":61,"benchmarks":8,"benchmark_tables_in_archive":8,"benchmark_tables_shown":8,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/point-tracking-on-tap-vid-davis","slug":"point-tracking-on-tap-vid-davis","dataset":"TAP-Vid-DAVIS","dataset_url":null,"rows_in_archive":3,"metrics":["Average Jaccard","Average PCK","Occlusion Accuracy"],"first_row_in_archive_order":{"model":"LocoTrack-B","paper_title":"Local All-Pair Correspondence for Point Tracking","paper_url":"/paper/local-all-pair-correspondence-for-point","paper_date":"2024-07-22","arxiv_id":"2407.15420","code_links":[{"title":"cvlab-kaist/locotrack","url":"https://github.com/cvlab-kaist/locotrack"},{"title":"ku-cvlab/locotrack","url":"https://github.com/ku-cvlab/locotrack"}],"syntology":{"n":12,"n_ran":6,"n_unverified":6,"n_pointer_only":0}}},{"leaderboard":"/sota/point-tracking-on-pointodyssey","slug":"point-tracking-on-pointodyssey","dataset":"PointOdyssey","dataset_url":"/dataset/pointodyssey","rows_in_archive":2,"metrics":["Survival","δ","MTE"],"first_row_in_archive_order":{"model":"PIPs++","paper_title":"PointOdyssey: A Large-Scale Synthetic Dataset for Long-Term Point Tracking","paper_url":"/paper/pointodyssey-a-large-scale-synthetic-dataset","paper_date":"2023-07-27","arxiv_id":"2307.15055","code_links":[{"title":"aharley/pips2","url":"https://github.com/aharley/pips2"},{"title":"y-zheng18/point_odyssey","url":"https://github.com/y-zheng18/point_odyssey"},{"title":"aliciachenw/PIPsUS","url":"https://github.com/aliciachenw/PIPsUS"}],"syntology":{"n":13,"n_ran":10,"n_unverified":3,"n_pointer_only":8}}},{"leaderboard":"/sota/point-tracking-on-tap-vid-davis-first","slug":"point-tracking-on-tap-vid-davis-first","dataset":"TAP-Vid-DAVIS-First","dataset_url":null,"rows_in_archive":2,"metrics":["Average Jaccard","Average PCK","Occlusion Accuracy"],"first_row_in_archive_order":{"model":"LocoTrack-B","paper_title":"Local All-Pair Correspondence for Point Tracking","paper_url":"/paper/local-all-pair-correspondence-for-point","paper_date":"2024-07-22","arxiv_id":"2407.15420","code_links":[{"title":"cvlab-kaist/locotrack","url":"https://github.com/cvlab-kaist/locotrack"},{"title":"ku-cvlab/locotrack","url":"https://github.com/ku-cvlab/locotrack"}],"syntology":{"n":12,"n_ran":6,"n_unverified":6,"n_pointer_only":0}}},{"leaderboard":"/sota/point-tracking-on-tap-vid-kinetics","slug":"point-tracking-on-tap-vid-kinetics","dataset":"TAP-Vid-Kinetics","dataset_url":null,"rows_in_archive":2,"metrics":["Average Jaccard","Average PCK","Occlusion Accuracy"],"first_row_in_archive_order":{"model":"BootsTAPIR","paper_title":"BootsTAP: Bootstrapped Training for Tracking-Any-Point","paper_url":"/paper/bootstap-bootstrapped-training-for-tracking","paper_date":"2024-02-01","arxiv_id":"2402.00847","code_links":[{"title":"google-deepmind/tapnet","url":"https://github.com/google-deepmind/tapnet"},{"title":"deepmind/tapnet","url":"https://github.com/deepmind/tapnet"}],"syntology":null}},{"leaderboard":"/sota/point-tracking-on-tap-vid-kinetics-first","slug":"point-tracking-on-tap-vid-kinetics-first","dataset":"TAP-Vid-Kinetics-First","dataset_url":null,"rows_in_archive":2,"metrics":["Average Jaccard","Average PCK","Occlusion Accuracy"],"first_row_in_archive_order":{"model":"LocoTrack-B","paper_title":"Local All-Pair Correspondence for Point Tracking","paper_url":"/paper/local-all-pair-correspondence-for-point","paper_date":"2024-07-22","arxiv_id":"2407.15420","code_links":[{"title":"cvlab-kaist/locotrack","url":"https://github.com/cvlab-kaist/locotrack"},{"title":"ku-cvlab/locotrack","url":"https://github.com/ku-cvlab/locotrack"}],"syntology":{"n":12,"n_ran":6,"n_unverified":6,"n_pointer_only":0}}},{"leaderboard":"/sota/point-tracking-on-tap-vid-rgb-stacking","slug":"point-tracking-on-tap-vid-rgb-stacking","dataset":"TAP-Vid-RGB-Stacking","dataset_url":null,"rows_in_archive":2,"metrics":["Average Jaccard","Average PCK","Occlusion Accuracy"],"first_row_in_archive_order":{"model":"BootsTAPIR","paper_title":"BootsTAP: Bootstrapped Training for Tracking-Any-Point","paper_url":"/paper/bootstap-bootstrapped-training-for-tracking","paper_date":"2024-02-01","arxiv_id":"2402.00847","code_links":[{"title":"google-deepmind/tapnet","url":"https://github.com/google-deepmind/tapnet"},{"title":"deepmind/tapnet","url":"https://github.com/deepmind/tapnet"}],"syntology":null}},{"leaderboard":"/sota/point-tracking-on-perception-test","slug":"point-tracking-on-perception-test","dataset":"Perception Test","dataset_url":"/dataset/perception-test","rows_in_archive":1,"metrics":["Average Jaccard"],"first_row_in_archive_order":{"model":"Static Baseline","paper_title":"Perception Test: A Diagnostic Benchmark for Multimodal Video Models","paper_url":"/paper/perception-test-a-diagnostic-benchmark-for-2","paper_date":"2023-05-23","arxiv_id":"2305.13786","code_links":[{"title":"deepmind/perception_test","url":"https://github.com/deepmind/perception_test"}],"syntology":null}},{"leaderboard":"/sota/point-tracking-on-tap-vid","slug":"point-tracking-on-tap-vid","dataset":"TAP-Vid","dataset_url":"/dataset/tap-vid","rows_in_archive":1,"metrics":["MTE","Survival","δ"],"first_row_in_archive_order":{"model":"PIPs++","paper_title":"PointOdyssey: A Large-Scale Synthetic Dataset for Long-Term Point Tracking","paper_url":"/paper/pointodyssey-a-large-scale-synthetic-dataset","paper_date":"2023-07-27","arxiv_id":"2307.15055","code_links":[{"title":"aharley/pips2","url":"https://github.com/aharley/pips2"},{"title":"y-zheng18/point_odyssey","url":"https://github.com/y-zheng18/point_odyssey"},{"title":"aliciachenw/PIPsUS","url":"https://github.com/aliciachenw/PIPsUS"}],"syntology":{"n":13,"n_ran":10,"n_unverified":3,"n_pointer_only":8}}}],"datasets":[{"url":"/dataset/tap-vid","name":"TAP-Vid","full_name":"","num_papers_in_archive":46},{"url":"/dataset/pointodyssey","name":"PointOdyssey","full_name":"","num_papers_in_archive":15},{"url":"/dataset/perception-test","name":"Perception Test","full_name":"","num_papers_in_archive":10},{"url":"/dataset/tapvid-3d-a-benchmark-for-tracking-any-point","name":"TAPVid-3D: A Benchmark for Tracking Any Point in 3D","full_name":"","num_papers_in_archive":2}],"subtasks":[],"parent_tasks":[{"url":"/task/visual-tracking","name":"Visual Tracking"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":61,"tagged_in_all":151,"items":[{"url":"/paper/drag-your-gan-interactive-point-based","title":"Drag Your GAN: Interactive Point-based Manipulation on the Generative Image Manifold","date":"2023-05-18","arxiv_id":"2305.10973","repositories_listed":7,"syntology":{"n":16,"n_ran":7,"n_unverified":9,"n_pointer_only":3}},{"url":"/paper/pointodyssey-a-large-scale-synthetic-dataset","title":"PointOdyssey: A Large-Scale Synthetic Dataset for Long-Term Point Tracking","date":"2023-07-27","arxiv_id":"2307.15055","repositories_listed":3,"syntology":{"n":13,"n_ran":10,"n_unverified":3,"n_pointer_only":8}},{"url":"/paper/tapir-tracking-any-point-with-per-frame","title":"TAPIR: Tracking Any Point with per-frame Initialization and temporal Refinement","date":"2023-06-14","arxiv_id":"2306.08637","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/tap-vid-a-benchmark-for-tracking-any-point-in","title":"TAP-Vid: A Benchmark for Tracking Any Point in a Video","date":"2022-11-07","arxiv_id":"2211.03726","repositories_listed":3,"syntology":{"n":12,"n_ran":4,"n_unverified":8,"n_pointer_only":2}},{"url":"/paper/frechet-video-motion-distance-a-metric-for","title":"Fréchet Video Motion Distance: A Metric for Evaluating Motion Consistency in Videos","date":"2024-07-23","arxiv_id":"2407.16124","repositories_listed":2,"syntology":{"n":23,"n_ran":20,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/local-all-pair-correspondence-for-point","title":"Local All-Pair Correspondence for Point Tracking","date":"2024-07-22","arxiv_id":"2407.15420","repositories_listed":2,"syntology":{"n":12,"n_ran":6,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/tapvid-3d-a-benchmark-for-tracking-any-point","title":"TAPVid-3D: A Benchmark for Tracking Any Point in 3D","date":"2024-07-08","arxiv_id":"2407.05921","repositories_listed":2,"syntology":null},{"url":"/paper/bootstap-bootstrapped-training-for-tracking","title":"BootsTAP: Bootstrapped Training for Tracking-Any-Point","date":"2024-02-01","arxiv_id":"2402.00847","repositories_listed":2,"syntology":null},{"url":"/paper/cotracker-it-is-better-to-track-together","title":"CoTracker: It is Better to Track Together","date":"2023-07-14","arxiv_id":"2307.07635","repositories_listed":2,"syntology":null},{"url":"/paper/spatialtrackerv2-3d-point-tracking-made-easy-1","title":"SpatialTrackerV2: 3D Point Tracking Made Easy","date":"2025-07-16","arxiv_id":"2507.12462","repositories_listed":1,"syntology":null},{"url":"/paper/characonsist-fine-grained-consistent","title":"CharaConsist: Fine-Grained Consistent Character Generation","date":"2025-07-15","arxiv_id":"2507.11533","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/draglora-online-optimization-of-lora-adapters","title":"DragLoRA: Online Optimization of LoRA Adapters for Drag-based Image Editing in Diffusion Model","date":"2025-05-18","arxiv_id":"2505.12427","repositories_listed":1,"syntology":null},{"url":"/paper/tapip3d-tracking-any-point-in-persistent-3d","title":"TAPIP3D: Tracking Any Point in Persistent 3D Geometry","date":"2025-04-20","arxiv_id":"2504.14717","repositories_listed":1,"syntology":{"n":8,"n_ran":2,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/seurat-from-moving-points-to-depth","title":"Seurat: From Moving Points to Depth","date":"2025-04-20","arxiv_id":"2504.14687","repositories_listed":1,"syntology":null},{"url":"/paper/pomato-marrying-pointmap-matching-with","title":"POMATO: Marrying Pointmap Matching with Temporal Motion for Dynamic 3D Reconstruction","date":"2025-04-08","arxiv_id":"2504.05692","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":6}},{"url":"/paper/tapnext-tracking-any-point-tap-as-next-token","title":"TAPNext: Tracking Any Point (TAP) as Next Token Prediction","date":"2025-04-08","arxiv_id":"2504.05579","repositories_listed":1,"syntology":null},{"url":"/paper/point-tracking-in-surgery-the-2024-surgical","title":"Point Tracking in Surgery--The 2024 Surgical Tattoos in Infrared (STIR) Challenge","date":"2025-03-31","arxiv_id":"2503.24306","repositories_listed":1,"syntology":null},{"url":"/paper/vggt-visual-geometry-grounded-transformer","title":"VGGT: Visual Geometry Grounded Transformer","date":"2025-03-14","arxiv_id":"2503.11651","repositories_listed":1,"syntology":null},{"url":"/paper/low-complexity-point-tracking-of-the","title":"Low Complexity Point Tracking of the Myocardium in 2D Echocardiography","date":"2025-03-13","arxiv_id":"2503.10431","repositories_listed":1,"syntology":null},{"url":"/paper/online-dense-point-tracking-with-streaming","title":"Online Dense Point Tracking with Streaming Memory","date":"2025-03-09","arxiv_id":"2503.06471","repositories_listed":1,"syntology":{"n":12,"n_ran":5,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/track-on-transformer-based-online-point","title":"Track-On: Transformer-based Online Point Tracking with Memory","date":"2025-01-30","arxiv_id":"2501.18487","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/exploring-temporally-aware-features-for-point","title":"Exploring Temporally-Aware Features for Point Tracking","date":"2025-01-21","arxiv_id":"2501.12218","repositories_listed":1,"syntology":null},{"url":"/paper/matcha-towards-matching-anything-1","title":"MATCHA: Towards Matching Anything","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cotracker3-simpler-and-better-point-tracking","title":"CoTracker3: Simpler and Better Point Tracking by Pseudo-Labelling Real Videos","date":"2024-10-15","arxiv_id":"2410.11831","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/self-supervised-any-point-tracking-by","title":"Self-Supervised Any-Point Tracking by Contrastive Random Walks","date":"2024-09-24","arxiv_id":"2409.16288","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/foundation-models-for-amodal-video-instance","title":"Foundation Models for Amodal Video Instance Segmentation in Automated Driving","date":"2024-09-21","arxiv_id":"2409.14095","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-object-priors-for-point-tracking","title":"Leveraging Object Priors for Point Tracking","date":"2024-09-09","arxiv_id":"2409.05786","repositories_listed":1,"syntology":null},{"url":"/paper/a-training-free-framework-for-video-license","title":"A Training-Free Framework for Video License Plate Tracking and Recognition with Only One-Shot","date":"2024-08-11","arxiv_id":"2408.05729","repositories_listed":1,"syntology":null},{"url":"/paper/autogenic-language-embedding-for-coherent","title":"Autogenic Language Embedding for Coherent Point Tracking","date":"2024-07-30","arxiv_id":"2407.20730","repositories_listed":1,"syntology":null},{"url":"/paper/motion-prior-contrast-maximization-for-dense","title":"Motion-prior Contrast Maximization for Dense Continuous-Time Motion Estimation","date":"2024-07-15","arxiv_id":"2407.10802","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}}],"syntology_records":14,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}