{"url":"/dataset/vrd","name":"VRD","full_name":"Visual Relationship Detection dataset","description_markdown":"The Visual Relationship Dataset (**VRD**) contains 4000 images for training and 1000 for testing annotated with visual relationships. Bounding boxes are annotated with a label containing 100 unary predicates. These labels refer to animals, vehicles, clothes and generic objects. Pairs of bounding boxes are annotated with a label containing 70 binary predicates. These labels refer to actions, prepositions, spatial relations, comparatives or preposition phrases. The dataset has 37993 instances of visual relationships and 6672 types of relationships. 1877 instances of relationships occur only in the test set and they are used to evaluate the zero-shot learning scenario.\r\n\r\nSource: [Compensating Supervision Incompleteness with Prior Knowledge in Semantic Image Interpretation](https://arxiv.org/abs/1910.00462)\r\nImage Source: [https://cs.stanford.edu/people/ranjaykrishna/vrd/](https://cs.stanford.edu/people/ranjaykrishna/vrd/)","description_withheld":null,"homepage":"https://cs.stanford.edu/people/ranjaykrishna/vrd/","introduced_date":"2016-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/visual-relationship-detection-with-language","title":"Visual Relationship Detection with Language Priors","first_author":"Cewu Lu","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Scene Graph Generation","url":"/task/scene-graph-generation","datasets_with_task":"/datasets/task/scene-graph-generation"},{"name":"Visual Relationship Detection","url":"/task/visual-relationship-detection","datasets_with_task":"/datasets/task/visual-relationship-detection"},{"name":"Scene Graph Detection","url":"/task/scene-graph-detection","datasets_with_task":"/datasets/task/scene-graph-detection"},{"name":"Relationship Detection","url":"/task/relationship-detection","datasets_with_task":"/datasets/task/relationship-detection"}],"languages":[],"variants":["VRD","VRD Predicate Detection","VRD Phrase Detection","VRD Relationship Detection"],"data_loaders":[],"num_papers_in_archive":151,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-relationship-detection-on-vrd-1","task":"Visual Relationship Detection","dataset_variant":"VRD Relationship Detection","rows":8,"metrics":["R@100","R@50"],"first_row_in_archive_order":{"model":"Yu et. al [[Yu et al.2017a]]","paper":"/paper/visual-relationship-detection-with-internal","metrics":{"R@100":"31.89","R@50":"22.68"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-relationship-detection-on-vrd","task":"Visual Relationship Detection","dataset_variant":"VRD Predicate Detection","rows":7,"metrics":["R@100","R@50"],"first_row_in_archive_order":{"model":"Yu et. al [[Yu et al.2017a]]","paper":"/paper/visual-relationship-detection-with-internal","metrics":{"R@100":"94.65","R@50":"85.64"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-relationship-detection-on-vrd-phrase","task":"Visual Relationship Detection","dataset_variant":"VRD Phrase Detection","rows":7,"metrics":["R@100","R@50"],"first_row_in_archive_order":{"model":"Yu et. al [[Yu et al.2017a]]","paper":"/paper/visual-relationship-detection-with-internal","metrics":{"R@100":"29.43","R@50":"26.32"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/scene-graph-generation-on-vrd","task":"Scene Graph Generation","dataset_variant":"VRD","rows":2,"metrics":["Recall@50"],"first_row_in_archive_order":{"model":"FactorizableNet","paper":"/paper/factorizable-net-an-efficient-subgraph-based","metrics":{"Recall@50":"18.32"},"code_links":[{"title":"yikang-li/FactorizableNet","url":"https://github.com/yikang-li/FactorizableNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-relationship-detection-on-vrd-2","task":"Visual Relationship Detection","dataset_variant":"VRD","rows":1,"metrics":["R@50 k=1"],"first_row_in_archive_order":{"model":"Ours - v","paper":"/paper/improving-visual-relation-detection-using","metrics":{"R@50 k=1":"15"},"code_links":[{"title":"Sina-Baharlou/Depth-VRD","url":"https://github.com/Sina-Baharlou/Depth-VRD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/improving-visual-relation-detection-using","title":"Improving Visual Relation Detection using Depth Maps","date":"2019-05-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/block-bilinear-superdiagonal-fusion-for","title":"BLOCK: Bilinear Superdiagonal Fusion for Visual Question Answering and Visual Relationship Detection","date":"2019-01-31","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/factorizable-net-an-efficient-subgraph-based","title":"Factorizable Net: An Efficient Subgraph-based Framework for Scene Graph Generation","date":"2018-06-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/visual-relationship-detection-with-deep","title":"Visual relationship detection with deep structural ranking","date":"2018-04-27","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/weakly-supervised-learning-of-visual","title":"Weakly-supervised learning of visual relations","date":"2017-07-29","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/visual-relationship-detection-with-internal","title":"Visual Relationship Detection with Internal and External Linguistic Knowledge Distillation","date":"2017-07-28","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/detecting-visual-relationships-with-deep","title":"Detecting Visual Relationships with Deep Relational Networks","date":"2017-04-11","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/deep-variation-structured-reinforcement","title":"Deep Variation-structured Reinforcement Learning for Visual Relationship and Attribute Detection","date":"2017-03-08","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/visual-translation-embedding-network-for","title":"Visual Translation Embedding Network for Visual Relation Detection","date":"2017-02-27","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/visual-relationship-detection-with-language","title":"Visual Relationship Detection with Language Priors","date":"2016-07-31","rows_on_this_dataset":4,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}