{"url":"/dataset/vidor","name":"VidOR","full_name":null,"description_markdown":"VidOR (Video Object Relation) dataset contains 10,000 videos (98.6 hours) from YFCC100M collection together with a large amount of fine-grained annotations for relation understanding. In particular, 80 categories of objects are annotated with bounding-box trajectory to indicate their spatio-temporal location in the videos; and 50 categories of relation predicates are annotated among all pairs of annotated objects with starting and ending frame index. This results in around 50,000 object and 380,000 relation instances annotated. To use the dataset for model development, the dataset is split into 7,000 videos for training, 835 videos for validation, and 2,165 videos for testing.","description_withheld":null,"homepage":"https://xdshang.github.io/docs/vidor.html","introduced_date":"2019-11-27","introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Video Visual Relation Detection","url":"/task/video-visual-relation-detection","datasets_with_task":"/datasets/task/video-visual-relation-detection"},{"name":"Video Visual Relation Tagging","url":"/task/video-visual-relation-tagging","datasets_with_task":"/datasets/task/video-visual-relation-tagging"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["VidOR"],"data_loaders":[],"num_papers_in_archive":5,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-visual-relation-detection-on-vidor","task":"Video Visual Relation Detection","dataset_variant":"VidOR","rows":2,"metrics":["Recall@100","Recall@50","mAP"],"first_row_in_archive_order":{"model":"Social Fabric","paper":"/paper/social-fabric-tubelet-compositions-for-video","metrics":{"Recall@100":"11.94","Recall@50":"9.99","mAP":"11.21"},"code_links":[{"title":"shanshuo/social-fabric","url":"https://github.com/shanshuo/social-fabric"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/social-fabric-tubelet-compositions-for-video","title":"Social Fabric: Tubelet Compositions for Video Relation Detection","date":"2021-08-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/what-and-when-to-look-temporal-span-proposal","title":"What and When to Look?: Temporal Span Proposal Network for Video Relation Detection","date":"2021-07-15","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}