{"url":"/task/video-object-detection","name":"Video Object Detection","slug":"video-object-detection","description_markdown":"Video object detection is the task of detecting objects from a video as opposed to images.\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [Learning Motion Priors for Efficient Video Object Detection](https://arxiv.org/pdf/1911.05253v1.pdf) )</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":147,"papers_with_code":72,"benchmarks":7,"benchmark_tables_in_archive":8,"benchmark_tables_shown":7,"benchmark_tables_withheld_as_spam":1,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":13,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/video-object-detection-on-imagenet-vid","slug":"video-object-detection-on-imagenet-vid","dataset":"ImageNet VID","dataset_url":"/dataset/imagenet-vid","rows_in_archive":33,"metrics":["MAP "],"first_row_in_archive_order":{"model":"YOLOV++","paper_title":"Practical Video Object Detection via Feature Selection and Aggregation","paper_url":"/paper/practical-video-object-detection-via-feature","paper_date":"2024-07-29","arxiv_id":"2407.19650","code_links":[{"title":"yuhengsss/yolov","url":"https://github.com/yuhengsss/yolov"}],"syntology":null}},{"leaderboard":"/sota/video-object-detection-on-epic-kitchens-1","slug":"video-object-detection-on-epic-kitchens-1","dataset":"EPIC KITCHENS-unseen splits","dataset_url":null,"rows_in_archive":1,"metrics":["mAP"],"first_row_in_archive_order":{"model":"Temporal ROI Align","paper_title":"Temporal RoI Align for Video Object Recognition","paper_url":"/paper/temporal-roi-align-for-video-object","paper_date":"2021-09-08","arxiv_id":"2109.03495","code_links":[{"title":"open-mmlab/mmtracking","url":"https://github.com/open-mmlab/mmtracking"}],"syntology":null}},{"leaderboard":"/sota/video-object-detection-on-epic-kitchens-55","slug":"video-object-detection-on-epic-kitchens-55","dataset":"EPIC-KITCHENS-55","dataset_url":"/dataset/epic-kitchens","rows_in_archive":1,"metrics":["mAP@.5"],"first_row_in_archive_order":{"model":"Ours (Faster RCNN)","paper_title":"Objects do not disappear: Video object detection by single-frame object location anticipation","paper_url":"/paper/objects-do-not-disappear-video-object","paper_date":"2023-08-09","arxiv_id":"2308.04770","code_links":[{"title":"l-kid/video-object-detection-by-location-anticipation","url":"https://github.com/l-kid/video-object-detection-by-location-anticipation"},{"title":"Elstuhn/Video-object-detection-by-location-anticipation","url":"https://github.com/Elstuhn/Video-object-detection-by-location-anticipation"}],"syntology":null}},{"leaderboard":"/sota/video-object-detection-on-epic-kitchens-seen","slug":"video-object-detection-on-epic-kitchens-seen","dataset":"EPIC KITCHENS-seen splits","dataset_url":null,"rows_in_archive":1,"metrics":["mAP"],"first_row_in_archive_order":{"model":"Temporal ROI Align","paper_title":"Temporal RoI Align for Video Object Recognition","paper_url":"/paper/temporal-roi-align-for-video-object","paper_date":"2021-09-08","arxiv_id":"2109.03495","code_links":[{"title":"open-mmlab/mmtracking","url":"https://github.com/open-mmlab/mmtracking"}],"syntology":null}},{"leaderboard":"/sota/video-object-detection-on-usc-grad-stddb","slug":"video-object-detection-on-usc-grad-stddb","dataset":"USC-GRAD-STDdb","dataset_url":"/dataset/usc-grad-stddb","rows_in_archive":1,"metrics":["AP","AP 0.5"],"first_row_in_archive_order":{"model":"SLTnet FPN-X101","paper_title":"Short-term anchor linking and long-term self-guided attention for video object detection","paper_url":"/paper/short-term-anchor-linking-and-long-term-self","paper_date":"2021-04-18","arxiv_id":null,"code_links":[{"title":"daniel-cores/SLTnet","url":"https://github.com/daniel-cores/SLTnet"}],"syntology":null}},{"leaderboard":"/sota/video-object-detection-on-waymo-open-dataset","slug":"video-object-detection-on-waymo-open-dataset","dataset":"Waymo Open Dataset","dataset_url":"/dataset/waymo-open-dataset","rows_in_archive":1,"metrics":["AP"],"first_row_in_archive_order":{"model":null,"paper_title":"Objects do not disappear: Video object detection by single-frame object location anticipation","paper_url":"/paper/objects-do-not-disappear-video-object","paper_date":"2023-08-09","arxiv_id":"2308.04770","code_links":[{"title":"l-kid/video-object-detection-by-location-anticipation","url":"https://github.com/l-kid/video-object-detection-by-location-anticipation"},{"title":"Elstuhn/Video-object-detection-by-location-anticipation","url":"https://github.com/Elstuhn/Video-object-detection-by-location-anticipation"}],"syntology":null}},{"leaderboard":"/sota/video-object-detection-on-yt-bb","slug":"video-object-detection-on-yt-bb","dataset":"YT-BB","dataset_url":"/dataset/youtube-boundingboxes","rows_in_archive":1,"metrics":["mAP"],"first_row_in_archive_order":{"model":null,"paper_title":"Objects do not disappear: Video object detection by single-frame object location anticipation","paper_url":"/paper/objects-do-not-disappear-video-object","paper_date":"2023-08-09","arxiv_id":"2308.04770","code_links":[{"title":"l-kid/video-object-detection-by-location-anticipation","url":"https://github.com/l-kid/video-object-detection-by-location-anticipation"},{"title":"Elstuhn/Video-object-detection-by-location-anticipation","url":"https://github.com/Elstuhn/Video-object-detection-by-location-anticipation"}],"syntology":null}}],"datasets":[{"url":"/dataset/waymo-open-dataset","name":"Waymo Open Dataset","full_name":"","num_papers_in_archive":481},{"url":"/dataset/epic-kitchens","name":"EPIC-KITCHENS-55","full_name":"","num_papers_in_archive":42},{"url":"/dataset/imagenet-vid","name":"ImageNet VID","full_name":"","num_papers_in_archive":26},{"url":"/dataset/gen1-detection","name":"GEN1 Detection","full_name":"Prophesee GEN1 Automotive Detection Dataset","num_papers_in_archive":13},{"url":"/dataset/dttd2","name":"DTTD-Mobile","full_name":"","num_papers_in_archive":8},{"url":"/dataset/youtube-boundingboxes","name":"YT-BB","full_name":"YouTube-BoundingBoxes","num_papers_in_archive":7},{"url":"/dataset/oak","name":"OAK","full_name":"Objects Around Krishna","num_papers_in_archive":5},{"url":"/dataset/synthia-al","name":"SYNTHIA-AL","full_name":"","num_papers_in_archive":5},{"url":"/dataset/5011-images-human-frontal-face-data-male","name":"5,011 Images – Human Frontal face Data (Male)","full_name":"5,011 Images – Human Frontal face Data (Male)","num_papers_in_archive":2},{"url":"/dataset/underwater-trash-detection","name":"Underwater Trash Detection","full_name":"","num_papers_in_archive":2},{"url":"/dataset/thgp","name":"THGP","full_name":"Temporal Hands Guns and Phones Dataset","num_papers_in_archive":1},{"url":"/dataset/usc-grad-stddb","name":"USC-GRAD-STDdb","full_name":"Small Target Detection database","num_papers_in_archive":1},{"url":"/dataset/visem-tracking","name":"VISEM-Tracking","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/object-detection","name":"Object Detection"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":72,"tagged_in_all":147,"items":[{"url":"/paper/emerging-properties-in-self-supervised-vision","title":"Emerging Properties in Self-Supervised Vision Transformers","date":"2021-04-29","arxiv_id":"2104.14294","repositories_listed":32,"syntology":{"n":20,"n_ran":5,"n_unverified":15,"n_pointer_only":2}},{"url":"/paper/temporal-shift-module-for-efficient-video","title":"TSM: Temporal Shift Module for Efficient Video Understanding","date":"2018-11-20","arxiv_id":"1811.08383","repositories_listed":13,"syntology":{"n":16,"n_ran":6,"n_unverified":10,"n_pointer_only":4}},{"url":"/paper/transvod-end-to-end-video-object-detection","title":"TransVOD: End-to-End Video Object Detection with Spatial-Temporal Transformers","date":"2022-01-13","arxiv_id":"2201.05047","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/houghnet-integrating-near-and-long-range-1","title":"HoughNet: Integrating near and long-range evidence for visual detection","date":"2021-04-14","arxiv_id":"2104.06773","repositories_listed":3,"syntology":null},{"url":"/paper/long-term-temporal-context-for-per-camera","title":"Context R-CNN: Long Term Temporal Context for Per-Camera Object Detection","date":"2019-12-07","arxiv_id":"1912.03538","repositories_listed":3,"syntology":null},{"url":"/paper/transferable-adversarial-attacks-for-image","title":"Transferable Adversarial Attacks for Image and Video Object Detection","date":"2018-11-30","arxiv_id":"1811.12641","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":5}},{"url":"/paper/towards-high-performance-video-object","title":"Towards High Performance Video Object Detection for Mobiles","date":"2018-04-16","arxiv_id":"1804.05830","repositories_listed":3,"syntology":null},{"url":"/paper/mobile-video-object-detection-with-temporally","title":"Mobile Video Object Detection with Temporally-Aware Feature Maps","date":"2017-11-17","arxiv_id":"1711.06368","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/spatio-temporal-prompting-network-for-robust-1","title":"Spatio-temporal Prompting Network for Robust Video Feature Extraction","date":"2024-02-04","arxiv_id":"2402.02574","repositories_listed":2,"syntology":null},{"url":"/paper/objects-do-not-disappear-video-object","title":"Objects do not disappear: Video object detection by single-frame object location anticipation","date":"2023-08-09","arxiv_id":"2308.04770","repositories_listed":2,"syntology":null},{"url":"/paper/instance-aware-context-focused-and-memory","title":"Instance-aware, Context-focused, and Memory-efficient Weakly Supervised Object Detection","date":"2020-04-09","arxiv_id":"2004.04725","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/memory-enhanced-global-local-aggregation-for","title":"Memory Enhanced Global-Local Aggregation for Video Object Detection","date":"2020-03-26","arxiv_id":"2003.12063","repositories_listed":2,"syntology":null},{"url":"/paper/plug-play-convolutional-regression-tracker","title":"Plug & Play Convolutional Regression Tracker for Video Object Detection","date":"2020-03-02","arxiv_id":"2003.00981","repositories_listed":2,"syntology":null},{"url":"/paper/vision-meets-drones-past-present-and-future","title":"Detection and Tracking Meet Drones Challenge","date":"2020-01-16","arxiv_id":"2001.06303","repositories_listed":2,"syntology":null},{"url":"/paper/relation-distillation-networks-for-video","title":"Relation Distillation Networks for Video Object Detection","date":"2019-08-26","arxiv_id":"1908.09511","repositories_listed":2,"syntology":null},{"url":"/paper/sequence-level-semantics-aggregation-for","title":"Sequence Level Semantics Aggregation for Video Object Detection","date":"2019-07-15","arxiv_id":"1907.06390","repositories_listed":2,"syntology":null},{"url":"/paper/looking-fast-and-slow-memory-guided-mobile","title":"Looking Fast and Slow: Memory-Guided Mobile Video Object Detection","date":"2019-03-25","arxiv_id":"1903.10172","repositories_listed":2,"syntology":null},{"url":"/paper/flow-guided-feature-aggregation-for-video","title":"Flow-Guided Feature Aggregation for Video Object Detection","date":"2017-03-29","arxiv_id":"1703.10025","repositories_listed":2,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":3}},{"url":"/paper/fade-a-dataset-for-detecting-falling-objects","title":"FADE: A Dataset for Detecting Falling Objects around Buildings in Video","date":"2024-08-11","arxiv_id":"2408.05750","repositories_listed":1,"syntology":null},{"url":"/paper/practical-video-object-detection-via-feature","title":"Practical Video Object Detection via Feature Selection and Aggregation","date":"2024-07-29","arxiv_id":"2407.19650","repositories_listed":1,"syntology":null},{"url":"/paper/multi-resolution-rescored-bytetrack-for-video","title":"Multi-resolution Rescored ByteTrack for Video Object Detection on Ultra-low-power Embedded Systems","date":"2024-04-17","arxiv_id":"2404.11488","repositories_listed":1,"syntology":null},{"url":"/paper/camera-clustering-for-scalable-stream-based","title":"Camera clustering for scalable stream-based active distillation","date":"2024-04-16","arxiv_id":"2404.10411","repositories_listed":1,"syntology":null},{"url":"/paper/detection-of-micromobility-vehicles-in-urban","title":"Detection of Micromobility Vehicles in Urban Traffic Videos","date":"2024-02-28","arxiv_id":"2402.18503","repositories_listed":1,"syntology":null},{"url":"/paper/stf-spatio-temporal-fusion-module-for","title":"STF: Spatio-Temporal Fusion Module for Improving Video Object Detection","date":"2024-02-16","arxiv_id":"2402.10752","repositories_listed":1,"syntology":null},{"url":"/paper/tdvit-temporal-dilated-video-transformer-for","title":"TDViT: Temporal Dilated Video Transformer for Dense Video Tasks","date":"2024-02-14","arxiv_id":"2402.09257","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-one-stage-video-object-detection-by","title":"Efficient One-stage Video Object Detection by Exploiting Temporal Consistency","date":"2024-02-14","arxiv_id":"2402.09241","repositories_listed":1,"syntology":null},{"url":"/paper/mamba-multi-level-aggregation-via-memory-bank","title":"MAMBA: Multi-level Aggregation via Memory Bank for Video Object Detection","date":"2024-01-18","arxiv_id":"2401.09923","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/diffusionvid-denoising-object-boxes-with","title":"DiffusionVID: Denoising Object Boxes with Spatio-temporal Conditioning for Video Object Detection","date":"2023-10-30","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/eventful-transformers-leveraging-temporal","title":"Eventful Transformers: Leveraging Temporal Redundancy in Vision Transformers","date":"2023-08-25","arxiv_id":"2308.13494","repositories_listed":1,"syntology":{"n":25,"n_ran":7,"n_unverified":18,"n_pointer_only":0}},{"url":"/paper/object-detection-difficulty-suppressing-over","title":"Object Detection Difficulty: Suppressing Over-aggregation for Faster and Better Video Object Detection","date":"2023-08-22","arxiv_id":"2308.11327","repositories_listed":1,"syntology":null}],"syntology_records":9,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}