{"url":"/dataset/imagenet-vidvrd","name":"ImageNet-VidVRD","full_name":null,"description_markdown":"ImageNet-VidVRD dataset contains 1,000 videos selected from ILVSRC2016-VID dataset based on whether the video contains clear visual relations. It is split into 800 training set and 200 test set, and covers common subject/objects of 35 categories and predicates of 132 categories. Ten people contributed to labeling the dataset, which includes object trajectory labeling and relation labeling. Since the ILVSRC2016-VID dataset has the object trajectory annotation for 30 categories already, we supplemented the annotations by labeling the remaining 5 categories. In order to save the labor of relation labeling, we labeled typical segments of the videos in the training set and the whole of the videos in the test set.","description_withheld":null,"homepage":"https://xdshang.github.io/docs/imagenet-vidvrd.html","introduced_date":"2017-10-23","introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Video Visual Relation Detection","url":"/task/video-visual-relation-detection","datasets_with_task":"/datasets/task/video-visual-relation-detection"},{"name":"Video Visual Relation Tagging","url":"/task/video-visual-relation-tagging","datasets_with_task":"/datasets/task/video-visual-relation-tagging"},{"name":"Video scene graph generation","url":"/task/video-scene-graph-generation","datasets_with_task":"/datasets/task/video-scene-graph-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["ImageNet-VidVRD"],"data_loaders":[],"num_papers_in_archive":10,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-visual-relation-detection-on-imagenet","task":"Video Visual Relation Detection","dataset_variant":"ImageNet-VidVRD","rows":2,"metrics":["Recall@100","Recall@50","mAP"],"first_row_in_archive_order":{"model":"Social Fabric","paper":"/paper/social-fabric-tubelet-compositions-for-video","metrics":{"Recall@100":"16.88","Recall@50":"13.73","mAP":"20.08"},"code_links":[{"title":"shanshuo/social-fabric","url":"https://github.com/shanshuo/social-fabric"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-scene-graph-generation-on-imagenet","task":"Video scene graph generation","dataset_variant":"ImageNet-VidVRD","rows":1,"metrics":["Recall@50"],"first_row_in_archive_order":{"model":"DLL","paper":"/paper/taking-a-closer-look-at-visual-relation","metrics":{"Recall@50":"14.13"},"code_links":[{"title":"Wq23333/DLL","url":"https://github.com/Wq23333/DLL"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/taking-a-closer-look-at-visual-relation","title":"Taking A Closer Look at Visual Relation: Unbiased Video Scene Graph Generation with Decoupled Label Learning","date":"2023-03-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/social-fabric-tubelet-compositions-for-video","title":"Social Fabric: Tubelet Compositions for Video Relation Detection","date":"2021-08-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/what-and-when-to-look-temporal-span-proposal","title":"What and When to Look?: Temporal Span Proposal Network for Video Relation Detection","date":"2021-07-15","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}