{"url":"/dataset/imagenet-vid","name":"ImageNet VID","full_name":null,"description_markdown":"ImageNet VID is a large-scale public dataset\r\nfor video object detection and contains more than 1M frames for training and\r\nmore than 100k frames for validation.","description_withheld":null,"homepage":"","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Video Object Detection","url":"/task/video-object-detection","datasets_with_task":"/datasets/task/video-object-detection"}],"languages":[],"variants":["ImageNet VID"],"data_loaders":[],"num_papers_in_archive":26,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-object-detection-on-imagenet-vid","task":"Video Object Detection","dataset_variant":"ImageNet VID","rows":33,"metrics":["MAP "],"first_row_in_archive_order":{"model":"YOLOV++","paper":"/paper/practical-video-object-detection-via-feature","metrics":{"MAP ":"93.2"},"code_links":[{"title":"yuhengsss/yolov","url":"https://github.com/yuhengsss/yolov"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/tgbformer-transformer-graphformer-blender","title":"TGBFormer: Transformer-GraphFormer Blender Network for Video Object Detection","date":"2025-03-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/practical-video-object-detection-via-feature","title":"Practical Video Object Detection via Feature Selection and Aggregation","date":"2024-07-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/diffusionvid-denoising-object-boxes-with","title":"DiffusionVID: Denoising Object Boxes with Spatio-temporal Conditioning for Video Object Detection","date":"2023-10-30","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/identity-consistent-aggregation-for-video","title":"Identity-Consistent Aggregation for Video Object Detection","date":"2023-08-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/objects-do-not-disappear-video-object","title":"Objects do not disappear: Video object detection by single-frame object location anticipation","date":"2023-08-09","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/boxmask-revisiting-bounding-box-supervision","title":"BoxMask: Revisiting Bounding Box Supervision for Video Object Detection","date":"2022-10-12","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/spatio-temporal-learnable-proposals-for-end","title":"Spatio-Temporal Learnable Proposals for End-to-End Video Object Detection","date":"2022-10-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ptseformer-progressive-temporal-spatial","title":"PTSEFormer: Progressive Temporal-Spatial Enhanced TransFormer Towards Video Object Detection","date":"2022-09-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":14,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dafa-diversity-aware-feature-aggregation-for","title":"DAFA: Diversity-Aware Feature Aggregation for Attention-Based Video Object Detection","date":"2022-09-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/yolov-making-still-image-object-detectors","title":"YOLOV: Making Still Image Object Detectors Great at Video Object Detection","date":"2022-08-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-sparse-transformer-with-attention","title":"Video Sparse Transformer With Attention-Guided Memory for Video Object Detection","date":"2022-06-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/transvod-end-to-end-video-object-detection","title":"TransVOD: End-to-End Video Object Detection with Spatial-Temporal Transformers","date":"2022-01-13","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/temporal-roi-align-for-video-object","title":"Temporal RoI Align for Video Object Recognition","date":"2021-09-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/short-term-anchor-linking-and-long-term-self","title":"Short-term anchor linking and long-term self-guided attention for video object detection","date":"2021-04-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/robust-and-efficient-post-processing-for","title":"Robust and Efficient Post-Processing for Video Object Detection (REPP)","date":"2020-10-01","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/mining-inter-video-proposal-relations-for","title":"Mining Inter-Video Proposal Relations for Video Object Detection","date":"2020-08-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/memory-enhanced-global-local-aggregation-for","title":"Memory Enhanced Global-Local Aggregation for Video Object Detection","date":"2020-03-26","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/learning-motion-priors-for-efficient-video","title":"Learning Where to Focus for Efficient Video Object Detection","date":"2019-11-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/sequence-level-semantics-aggregation-for","title":"Sequence Level Semantics Aggregation for Video Object Detection","date":"2019-07-15","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/looking-fast-and-slow-memory-guided-mobile","title":"Looking Fast and Slow: Memory-Guided Mobile Video Object Detection","date":"2019-03-25","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/integrated-object-detection-and-tracking-with","title":"Integrated Object Detection and Tracking with Tracklet-Conditioned Detection","date":"2018-11-27","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/temporal-shift-module-for-efficient-video","title":"TSM: Temporal Shift Module for Efficient Video Understanding","date":"2018-11-20","rows_on_this_dataset":1,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":6,"samples_unverified":10,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/flow-guided-feature-aggregation-for-video","title":"Flow-Guided Feature Aggregation for Video Object Detection","date":"2017-03-29","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":46,"samples_ran":24,"samples_unverified":22,"pointer_only_for_licence":10,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}