{"url":"/task/video-object-tracking","name":"Video Object Tracking","slug":"video-object-tracking","description_markdown":"Video Object Detection aims to detect targets in videos using both spatial and temporal information. It's usually deeply integrated with tasks such as Object Detection and Object Tracking.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":98,"papers_with_code":72,"benchmarks":4,"benchmark_tables_in_archive":4,"benchmark_tables_shown":4,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":13,"subtasks":0,"parent_tasks":2},"benchmarks":[{"leaderboard":"/sota/video-object-tracking-on-nv-vot211","slug":"video-object-tracking-on-nv-vot211","dataset":"NT-VOT211","dataset_url":"/dataset/nv-vot211","rows_in_archive":43,"metrics":["AUC","Precision"],"first_row_in_archive_order":{"model":"ProContEXT","paper_title":"ProContEXT: Exploring Progressive Context Transformer for Tracking","paper_url":"/paper/procontext-exploring-progressive-context","paper_date":"2022-10-27","arxiv_id":"2210.15511","code_links":[{"title":"jp-lan/procontext","url":"https://github.com/jp-lan/procontext"},{"title":"yangyucheng000/Paper-4","url":"https://github.com/yangyucheng000/Paper-4/tree/main/ProC-KD-main"},{"title":"zhiqic/procontext","url":"https://github.com/zhiqic/procontext"},{"title":"yangyucheng000/papercode-2","url":"https://github.com/yangyucheng000/papercode-2/tree/main/ProC-KD-main"}],"syntology":null}},{"leaderboard":"/sota/video-object-tracking-on-cater","slug":"video-object-tracking-on-cater","dataset":"CATER","dataset_url":"/dataset/cater","rows_in_archive":7,"metrics":["Top 1 Accuracy","L1","Top 5 Accuracy"],"first_row_in_archive_order":{"model":"Loci","paper_title":"Learning What and Where: Disentangling Location and Identity Tracking Without Supervision","paper_url":"/paper/learning-what-and-where-unsupervised","paper_date":"2022-05-26","arxiv_id":"2205.13349","code_links":[{"title":"CognitiveModeling/Loci","url":"https://github.com/CognitiveModeling/Loci"}],"syntology":null}},{"leaderboard":"/sota/video-object-tracking-on-got-10k-1","slug":"video-object-tracking-on-got-10k-1","dataset":"GOT-10k","dataset_url":"/dataset/got-10k","rows_in_archive":1,"metrics":["Average Overlap"],"first_row_in_archive_order":{"model":"TATrack-L-GOT","paper_title":"Target-Aware Tracking with Long-term Context Attention","paper_url":"/paper/target-aware-tracking-with-long-term-context","paper_date":"2023-02-27","arxiv_id":"2302.13840","code_links":[{"title":"hekaijie123/TATrack","url":"https://github.com/hekaijie123/TATrack"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/video-object-tracking-on-soccernet-v2","slug":"video-object-tracking-on-soccernet-v2","dataset":"SoccerNet-v2","dataset_url":"/dataset/soccernet-v2","rows_in_archive":1,"metrics":["HOTA"],"first_row_in_archive_order":{"model":"CO-MOT","paper_title":"Bridging the Gap Between End-to-end and Non-End-to-end Multi-Object Tracking","paper_url":"/paper/bridging-the-gap-between-end-to-end-and-non","paper_date":"2023-05-22","arxiv_id":"2305.12724","code_links":[{"title":"bingfengyan/visam","url":"https://github.com/bingfengyan/visam"},{"title":"BingfengYan/CO-MOT","url":"https://github.com/BingfengYan/CO-MOT"}],"syntology":null}}],"datasets":[{"url":"/dataset/got-10k","name":"GOT-10k","full_name":"Generic Object Tracking Benchmark","num_papers_in_archive":239},{"url":"/dataset/soccernet-v2","name":"SoccerNet-v2","full_name":"","num_papers_in_archive":58},{"url":"/dataset/cater","name":"CATER","full_name":"","num_papers_in_archive":51},{"url":"/dataset/nv-vot211","name":"NT-VOT211","full_name":"","num_papers_in_archive":41},{"url":"/dataset/vot","name":"VOTChallenge","full_name":"Visual Object Tracking","num_papers_in_archive":36},{"url":"/dataset/vot2014","name":"VOT2014","full_name":"Visual Object Tracking Challenge 2014","num_papers_in_archive":12},{"url":"/dataset/bl30k","name":"BL30K","full_name":"","num_papers_in_archive":11},{"url":"/dataset/dttd2","name":"DTTD-Mobile","full_name":"","num_papers_in_archive":8},{"url":"/dataset/trek150","name":"TREK-150","full_name":"","num_papers_in_archive":7},{"url":"/dataset/videocube","name":"VideoCube","full_name":"","num_papers_in_archive":6},{"url":"/dataset/rf100","name":"RF100","full_name":"Roboflow 100","num_papers_in_archive":5},{"url":"/dataset/sotverse","name":"SOTVerse","full_name":"","num_papers_in_archive":1},{"url":"/dataset/visem-tracking","name":"VISEM-Tracking","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/object-tracking","name":"Object Tracking"},{"url":"/task/video","name":"Video"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":72,"tagged_in_all":98,"items":[{"url":"/paper/quo-vadis-action-recognition-a-new-model-and","title":"Quo Vadis, Action Recognition? A New Model and the Kinetics Dataset","date":"2017-05-22","arxiv_id":"1705.07750","repositories_listed":34,"syntology":{"n":28,"n_ran":16,"n_unverified":12,"n_pointer_only":7}},{"url":"/paper/yolov7-trainable-bag-of-freebies-sets-new","title":"YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors","date":"2022-07-06","arxiv_id":"2207.02696","repositories_listed":21,"syntology":{"n":11,"n_ran":1,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/temporal-shift-module-for-efficient-video","title":"TSM: Temporal Shift Module for Efficient Video Understanding","date":"2018-11-20","arxiv_id":"1811.08383","repositories_listed":13,"syntology":{"n":16,"n_ran":6,"n_unverified":10,"n_pointer_only":4}},{"url":"/paper/high-speed-tracking-with-kernelized","title":"High-Speed Tracking with Kernelized Correlation Filters","date":"2014-04-30","arxiv_id":"1404.7584","repositories_listed":9,"syntology":null},{"url":"/paper/deeper-and-wider-siamese-networks-for-real","title":"Deeper and Wider Siamese Networks for Real-Time Visual Tracking","date":"2019-01-07","arxiv_id":"1901.01660","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/fully-convolutional-siamese-networks-for","title":"Fully Convolutional Siamese Networks for Change Detection","date":"2018-10-19","arxiv_id":"1810.08462","repositories_listed":5,"syntology":null},{"url":"/paper/high-performance-visual-tracking-with-siamese","title":"High Performance Visual Tracking With Siamese Region Proposal Network","date":"2018-06-01","arxiv_id":null,"repositories_listed":5,"syntology":null},{"url":"/paper/procontext-exploring-progressive-context","title":"ProContEXT: Exploring Progressive Context Transformer for Tracking","date":"2022-10-27","arxiv_id":"2210.15511","repositories_listed":4,"syntology":null},{"url":"/paper/video-polyp-segmentation-a-deep-learning","title":"Video Polyp Segmentation: A Deep Learning Perspective","date":"2022-03-27","arxiv_id":"2203.14291","repositories_listed":4,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/ocean-object-aware-anchor-free-tracking","title":"Ocean: Object-aware Anchor-free Tracking","date":"2020-06-18","arxiv_id":"2006.10721","repositories_listed":4,"syntology":null},{"url":"/paper/fast-online-object-tracking-and-segmentation","title":"Fast Online Object Tracking and Segmentation: A Unifying Approach","date":"2018-12-12","arxiv_id":"1812.05050","repositories_listed":3,"syntology":{"n":10,"n_ran":1,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/staple-complementary-learners-for-real-time","title":"Staple: Complementary Learners for Real-Time Tracking","date":"2015-12-04","arxiv_id":"1512.01355","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/2408-00874","title":"Medical SAM 2: Segment medical images as video via Segment Anything Model 2","date":"2024-08-01","arxiv_id":"2408.00874","repositories_listed":2,"syntology":{"n":11,"n_ran":7,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/towards-a-generalist-and-blind-rgb-x-tracker","title":"XTrack: Multimodal Training Boosts RGB-X Video Object Trackers","date":"2024-05-28","arxiv_id":"2405.17773","repositories_listed":2,"syntology":null},{"url":"/paper/bridging-the-gap-between-end-to-end-and-non","title":"Bridging the Gap Between End-to-end and Non-End-to-end Multi-Object Tracking","date":"2023-05-22","arxiv_id":"2305.12724","repositories_listed":2,"syntology":null},{"url":"/paper/towards-sequence-level-training-for-visual","title":"Towards Sequence-Level Training for Visual Tracking","date":"2022-08-11","arxiv_id":"2208.05810","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":5}},{"url":"/paper/robust-visual-tracking-by-segmentation","title":"Robust Visual Tracking by Segmentation","date":"2022-03-21","arxiv_id":"2203.11191","repositories_listed":2,"syntology":null},{"url":"/paper/efficient-visual-tracking-with-exemplar","title":"Efficient Visual Tracking with Exemplar Transformers","date":"2021-12-17","arxiv_id":"2112.09686","repositories_listed":2,"syntology":null},{"url":"/paper/cater-a-diagnostic-dataset-for-compositional","title":"CATER: A diagnostic dataset for Compositional Actions and TEmporal Reasoning","date":"2019-10-10","arxiv_id":"1910.04744","repositories_listed":2,"syntology":{"n":12,"n_ran":0,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/190407220","title":"Learning Discriminative Model Prediction for Tracking","date":"2019-04-15","arxiv_id":"1904.07220","repositories_listed":2,"syntology":null},{"url":"/paper/robust-estimation-of-similarity","title":"Robust Estimation of Similarity Transformation for Visual Object Tracking","date":"2017-12-14","arxiv_id":"1712.05231","repositories_listed":2,"syntology":null},{"url":"/paper/him2sam-enhancing-sam2-with-hierarchical","title":"HiM2SAM: Enhancing SAM2 with Hierarchical Motion Estimation and Memory Optimization towards Long-term Tracking","date":"2025-07-10","arxiv_id":"2507.07603","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-multimodal-spatial-temporal","title":"Exploiting Multimodal Spatial-temporal Patterns for Video Object Tracking","date":"2024-12-20","arxiv_id":"2412.15691","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/exploring-enhanced-contextual-information-for-1","title":"Exploring Enhanced Contextual Information for Video-Level Object Tracking","date":"2024-12-15","arxiv_id":"2412.11023","repositories_listed":1,"syntology":null},{"url":"/paper/referring-video-object-segmentation-via","title":"Referring Video Object Segmentation via Language-aligned Track Selection","date":"2024-12-02","arxiv_id":"2412.01136","repositories_listed":1,"syntology":null},{"url":"/paper/teaching-vlms-to-localize-specific-objects","title":"Teaching VLMs to Localize Specific Objects from In-context Examples","date":"2024-11-20","arxiv_id":"2411.13317","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":4}},{"url":"/paper/nt-vot211-a-large-scale-benchmark-for-night","title":"NT-VOT211: A Large-Scale Benchmark for Night-time Visual Object Tracking","date":"2024-10-27","arxiv_id":"2410.20421","repositories_listed":1,"syntology":null},{"url":"/paper/depth-attention-for-robust-rgb-tracking","title":"Depth Attention for Robust RGB Tracking","date":"2024-10-27","arxiv_id":"2410.20395","repositories_listed":1,"syntology":null},{"url":"/paper/associate-everything-detected-facilitating","title":"Associate Everything Detected: Facilitating Tracking-by-Detection to the Unknown","date":"2024-09-14","arxiv_id":"2409.09293","repositories_listed":1,"syntology":null},{"url":"/paper/univs-unified-and-universal-video","title":"UniVS: Unified and Universal Video Segmentation with Prompts as Queries","date":"2024-02-28","arxiv_id":"2402.18115","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_unverified":2,"n_pointer_only":14}}],"syntology_records":13,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}