{"url":"/dataset/visual-genome","name":"Visual Genome","full_name":null,"description_markdown":"**Visual Genome** contains Visual Question Answering data in a multi-choice setting. It consists of 101,174 images from MSCOCO with 1.7 million QA pairs, 17 questions per image on average. Compared to the Visual Question Answering dataset, Visual Genome represents a more balanced distribution over 6 question types: What, Where, When, Who, Why and How. The Visual Genome dataset also presents 108K images with densely annotated objects, attributes and relationships.\r\n\r\nSource: [RaAM: A Relation-aware Attention Model for Visual Question Answering](https://arxiv.org/abs/1903.12314)\r\nImage Source: [Visual Genome: Connecting Language and Vision Using Crowdsourced Dense Image Annotations](https://paperswithcode.com/paper/visual-genome-connecting-language-and-vision/)","description_withheld":null,"homepage":"https://homes.cs.washington.edu/~ranjay/visualgenome/index.html","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/visual-genome-connecting-language-and-vision","title":"Visual Genome: Connecting Language and Vision Using Crowdsourced Dense Image Annotations","first_author":"Ranjay Krishna","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Object Detection","url":"/task/object-detection","datasets_with_task":"/datasets/task/object-detection"},{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Layout-to-Image Generation","url":"/task/layout-to-image-generation","datasets_with_task":"/datasets/task/layout-to-image-generation"},{"name":"Scene Graph Generation","url":"/task/scene-graph-generation","datasets_with_task":"/datasets/task/scene-graph-generation"},{"name":"Visual Relationship Detection","url":"/task/visual-relationship-detection","datasets_with_task":"/datasets/task/visual-relationship-detection"},{"name":"Phrase Grounding","url":"/task/phrase-grounding","datasets_with_task":"/datasets/task/phrase-grounding"},{"name":"Unsupervised KG-to-Text Generation","url":"/task/unsupervised-kg-to-text-generation","datasets_with_task":"/datasets/task/unsupervised-kg-to-text-generation"},{"name":"Scene Graph Detection","url":"/task/scene-graph-detection","datasets_with_task":"/datasets/task/scene-graph-detection"},{"name":"Predicate Classification","url":"/task/predicate-classification","datasets_with_task":"/datasets/task/predicate-classification"},{"name":"Image Generation from Scene Graphs","url":"/task/image-generation-from-scene-graphs","datasets_with_task":"/datasets/task/image-generation-from-scene-graphs"},{"name":"Multi-label Image Recognition with Partial Labels","url":"/task/multi-label-image-recognition-with-partial","datasets_with_task":"/datasets/task/multi-label-image-recognition-with-partial"},{"name":"Scene Graph Classification","url":"/task/scene-graph-classification","datasets_with_task":"/datasets/task/scene-graph-classification"},{"name":"Unsupervised semantic parsing","url":"/task/unsupervised-semantic-parsing","datasets_with_task":"/datasets/task/unsupervised-semantic-parsing"},{"name":"Dense Captioning","url":"/task/dense-captioning","datasets_with_task":"/datasets/task/dense-captioning"},{"name":"Bidirectional Relationship Classification","url":"/task/bidirectional-relationship-classification","datasets_with_task":"/datasets/task/bidirectional-relationship-classification"},{"name":"Unbiased Scene Graph Generation","url":"/task/unbiased-scene-graph-generation","datasets_with_task":"/datasets/task/unbiased-scene-graph-generation"}],"languages":[],"variants":["VG graph-text","Visual Genome 256x256","Visual Genome 64x64","Visual Genome 128x128","Visual Genome (subjects)","Visual Genome (pairs)","Visual Genome"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/ranjaykrishna/visual_genome","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/visual_genome","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":1256,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/unbiased-scene-graph-generation-on-visual","task":"Unbiased Scene Graph Generation","dataset_variant":"Visual Genome","rows":31,"metrics":["ng-mR@20","mR@20","F@100"],"first_row_in_archive_order":{"model":"IETrans (MOTIFS-ResNeXt-101-FPN backbone; PredCls mode)","paper":"/paper/fine-grained-scene-graph-generation-with-data","metrics":{"F@100":"44.1","mR@20":"28.9","ng-mR@20":"36.0"},"code_links":[{"title":"waxnkw/ietrans-sgg.pytorch","url":"https://github.com/waxnkw/ietrans-sgg.pytorch"},{"title":"rlqja1107/torch-st-sgg","url":"https://github.com/rlqja1107/torch-st-sgg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/scene-graph-generation-on-visual-genome","task":"Scene Graph Generation","dataset_variant":"Visual Genome","rows":19,"metrics":["Recall@50","mean Recall @20","Recall@100","Recall@20","mean Recall @100","R@100","mR@100","mR@50","zR@100","zR@20","zR@50"],"first_row_in_archive_order":{"model":"SpeaQ (without reweighting)","paper":"/paper/groupwise-query-specialization-and-quality","metrics":{"R@100":"36.0","Recall@100":"36.0","Recall@50":"32.9","mR@100":"14.1","mR@50":"11.8","mean Recall @100":"14.1"},"code_links":[{"title":"mlvlab/speaq","url":"https://github.com/mlvlab/speaq"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/layout-to-image-generation-on-visual-genome-3","task":"Layout-to-Image Generation","dataset_variant":"Visual Genome 128x128","rows":5,"metrics":["FID","Inception Score","SceneFID"],"first_row_in_archive_order":{"model":"LayoutDiffusion","paper":"/paper/layoutdiffusion-controllable-diffusion-model","metrics":{"FID":"16.35"},"code_links":[{"title":"zgctroy/layoutdiffusion","url":"https://github.com/zgctroy/layoutdiffusion"},{"title":"dcdcvgroup/layout-diffusion-mindspore","url":"https://github.com/dcdcvgroup/layout-diffusion-mindspore"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/dense-captioning-on-visual-genome","task":"Dense Captioning","dataset_variant":"Visual Genome","rows":4,"metrics":["mAP"],"first_row_in_archive_order":{"model":"ControlCap","paper":"/paper/controllable-dense-captioner-with-multimodal","metrics":{"mAP":"18.2"},"code_links":[{"title":"callsys/controlcap","url":"https://github.com/callsys/controlcap"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/layout-to-image-generation-on-visual-genome-2","task":"Layout-to-Image Generation","dataset_variant":"Visual Genome 64x64","rows":4,"metrics":["FID","Inception Score"],"first_row_in_archive_order":{"model":"OC-GAN","paper":"/paper/object-centric-image-generation-from-layouts","metrics":{"FID":"20.27","Inception Score":"9.3"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/layout-to-image-generation-on-visual-genome-4","task":"Layout-to-Image Generation","dataset_variant":"Visual Genome 256x256","rows":4,"metrics":["FID","Inception Score","LPIPS"],"first_row_in_archive_order":{"model":"LayoutDiffusion","paper":"/paper/layoutdiffusion-controllable-diffusion-model","metrics":{"FID":"15.63"},"code_links":[{"title":"zgctroy/layoutdiffusion","url":"https://github.com/zgctroy/layoutdiffusion"},{"title":"dcdcvgroup/layout-diffusion-mindspore","url":"https://github.com/dcdcvgroup/layout-diffusion-mindspore"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-label-image-recognition-with-partial-2","task":"Multi-label Image Recognition with Partial Labels","dataset_variant":"Visual Genome","rows":4,"metrics":["Average mAP"],"first_row_in_archive_order":{"model":"DSRB","paper":"/paper/semantic-aware-representation-blending-for-1","metrics":{"Average mAP":"46"},"code_links":[{"title":"hcplab-sysu/hcp-mlr-pl","url":"https://github.com/hcplab-sysu/hcp-mlr-pl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-detection-on-visual-genome","task":"Object Detection","dataset_variant":"Visual Genome","rows":4,"metrics":["MAP"],"first_row_in_archive_order":{"model":"KnowZRel","paper":"/paper/knowzrel-common-sense-knowledge-based-zero","metrics":{"MAP":"44"},"code_links":[{"title":"jaleedkhan/zsrr-sgg","url":"https://github.com/jaleedkhan/zsrr-sgg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/phrase-grounding-on-visual-genome","task":"Phrase Grounding","dataset_variant":"Visual Genome","rows":3,"metrics":["Pointing Game Accuracy"],"first_row_in_archive_order":{"model":"GbS VG","paper":"/paper/detector-free-weakly-supervised-grounding-by","metrics":{"Pointing Game Accuracy":"55.91"},"code_links":[{"title":"aarbelle/GroundingBySeparation","url":"https://github.com/aarbelle/GroundingBySeparation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-generation-from-scene-graphs-on-visual","task":"Image Generation from Scene Graphs","dataset_variant":"Visual Genome 64x64","rows":2,"metrics":["FID"],"first_row_in_archive_order":{"model":"MIGS","paper":"/paper/migs-meta-image-generation-from-scene-graphs","metrics":{"FID":"54.24"},"code_links":[{"title":"migs2021/migs","url":"https://github.com/migs2021/migs"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-kg-to-text-generation-on-vg","task":"Unsupervised KG-to-Text Generation","dataset_variant":"VG graph-text","rows":1,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"GT-BT (composed noise)","paper":"/paper/unsupervised-text-generation-from-structured","metrics":{"BLEU":"23.2"},"code_links":[{"title":"mnschmit/unsupervised-graph-text-conversion","url":"https://github.com/mnschmit/unsupervised-graph-text-conversion"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-semantic-parsing-on-vg-graph","task":"Unsupervised semantic parsing","dataset_variant":"VG graph-text","rows":1,"metrics":["F1"],"first_row_in_archive_order":{"model":"GT-BT (composed noise)","paper":"/paper/unsupervised-text-generation-from-structured","metrics":{"F1":"21.7"},"code_links":[{"title":"mnschmit/unsupervised-graph-text-conversion","url":"https://github.com/mnschmit/unsupervised-graph-text-conversion"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-visual-genome","task":"Visual Question Answering (VQA)","dataset_variant":"Visual Genome (subjects)","rows":1,"metrics":["Percentage correct"],"first_row_in_archive_order":{"model":"CMN","paper":"/paper/modeling-relationships-in-referential","metrics":{"Percentage correct":"44.24"},"code_links":[{"title":"hengyuan-hu/bottom-up-attention-vqa","url":"https://github.com/hengyuan-hu/bottom-up-attention-vqa"},{"title":"thilinicooray/Bottom-up-vqa","url":"https://github.com/thilinicooray/Bottom-up-vqa"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-visual-genome-1","task":"Visual Question Answering (VQA)","dataset_variant":"Visual Genome (pairs)","rows":1,"metrics":["Percentage correct"],"first_row_in_archive_order":{"model":"CMN","paper":"/paper/modeling-relationships-in-referential","metrics":{"Percentage correct":"28.52"},"code_links":[{"title":"hengyuan-hu/bottom-up-attention-vqa","url":"https://github.com/hengyuan-hu/bottom-up-attention-vqa"},{"title":"thilinicooray/Bottom-up-vqa","url":"https://github.com/thilinicooray/Bottom-up-vqa"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-relationship-detection-on-visual","task":"Visual Relationship Detection","dataset_variant":"Visual Genome","rows":1,"metrics":["R@100","R@50","mR@100","mR@50"],"first_row_in_archive_order":{"model":"PEVL","paper":"/paper/pevl-position-enhanced-pre-training-and","metrics":{"R@100":"66.3","R@50":"64.4","mR@100":"23.5","mR@50":"21.7"},"code_links":[{"title":"thunlp/pevl","url":"https://github.com/thunlp/pevl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/knowzrel-common-sense-knowledge-based-zero","title":"KnowZRel: Common Sense Knowledge-based Zero-Shot Relationship Retrieval for Generalised Scene Graph Generation","date":"2025-02-21","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/semantic-diversity-aware-prototype-based","title":"Semantic Diversity-aware Prototype-based Learning for Unbiased Scene Graph Generation","date":"2024-07-22","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/groupwise-query-specialization-and-quality","title":"Groupwise Query Specialization and Quality-Aware Multi-Assignment for Transformer-based Visual Relationship Detection","date":"2024-03-26","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/controllable-dense-captioner-with-multimodal","title":"ControlCap: Controllable Region-level Captioning","date":"2024-01-31","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/neusyre-neuro-symbolic-visual-understanding","title":"NeuSyRE: Neuro-Symbolic Visual Understanding and Reasoning Framework based on Scene Graph Enrichment","date":"2023-11-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/panoptic-scene-graph-generation-with","title":"Panoptic Scene Graph Generation with Semantics-Prototype Learning","date":"2023-07-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/layoutdiffusion-controllable-diffusion-model","title":"LayoutDiffusion: Controllable Diffusion Model for Layout-to-image Generation","date":"2023-03-30","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":5,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/grit-a-generative-region-to-text-transformer","title":"GRiT: A Generative Region-to-text Transformer for Object Understanding","date":"2022-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-enhanced-object-detection-model-for-scene","title":"An Enhanced Object Detection Model for Scene Graph Generation","date":"2022-11-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/expressive-scene-graph-generation-using","title":"Expressive Scene Graph Generation Using Commonsense Knowledge Infusion for Visual Understanding and Reasoning","date":"2022-05-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/semantic-aware-representation-blending-for-1","title":"Dual-Perspective Semantic-Aware Representation Blending for Multi-Label Image Recognition with Partial Labels","date":"2022-05-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pevl-position-enhanced-pre-training-and","title":"PEVL: Position-enhanced Pre-training and Prompt Tuning for Vision-language Models","date":"2022-05-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":3,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/heterogeneous-semantic-transfer-for-multi","title":"Heterogeneous Semantic Transfer for Multi-label Recognition with Partial Labels","date":"2022-05-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/fine-grained-scene-graph-generation-with-data","title":"Fine-Grained Scene Graph Generation with Data Transfer","date":"2022-03-22","rows_on_this_dataset":4,"code_links":2,"syntology":null},{"paper":"/paper/stacked-hybrid-attention-and-group","title":"Stacked Hybrid-Attention and Group Collaborative Learning for Unbiased Scene Graph Generation","date":"2022-03-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":3,"samples_unverified":4,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biasing-like-human-a-cognitive-bias-framework","title":"Biasing Like Human: A Cognitive Bias Framework for Scene Graph Generation","date":"2022-03-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/semantic-aware-representation-blending-for","title":"Semantic-Aware Representation Blending for Multi-Label Image Recognition with Partial Labels","date":"2022-03-04","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/interactive-image-synthesis-with-panoptic","title":"Interactive Image Synthesis with Panoptic Layout Generation","date":"2022-03-04","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":0,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/structured-semantic-transfer-for-multi-label","title":"Structured Semantic Transfer for Multi-Label Recognition with Partial Labels","date":"2021-12-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/migs-meta-image-generation-from-scene-graphs","title":"MIGS: Meta Image Generation from Scene Graphs","date":"2021-10-22","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/recovering-the-unbiased-scene-graphs-from-the","title":"Recovering the Unbiased Scene Graphs from the Biased Ones","date":"2021-07-05","rows_on_this_dataset":7,"code_links":1,"syntology":null},{"paper":"/paper/tackling-the-challenges-in-scene-graph","title":"Tackling the Challenges in Scene Graph Generation with Local-to-Global Interactions","date":"2021-06-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/detector-free-weakly-supervised-grounding-by","title":"Detector-Free Weakly Supervised Grounding by Separation","date":"2021-04-20","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/energy-based-learning-for-scene-graph","title":"Energy-Based Learning for Scene Graph Generation","date":"2021-03-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/cogtree-cognition-tree-loss-for-unbiased","title":"CogTree: Cognition Tree Loss for Unbiased Scene Graph Generation","date":"2020-09-16","rows_on_this_dataset":7,"code_links":1,"syntology":null},{"paper":"/paper/pcpl-predicate-correlation-perception","title":"PCPL: Predicate-Correlation Perception Learning for Unbiased Scene Graph Generation","date":"2020-09-02","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/gps-net-graph-property-sensing-network-for","title":"GPS-Net: Graph Property Sensing Network for Scene Graph Generation","date":"2020-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":2,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-layout-and-style-reconfigurable-gans","title":"Learning Layout and Style Reconfigurable GANs for Controllable Image Synthesis","date":"2020-03-25","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/object-centric-image-generation-from-layouts","title":"Object-Centric Image Generation from Layouts","date":"2020-03-16","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/unbiased-scene-graph-generation-from-biased","title":"Unbiased Scene Graph Generation from Biased Training","date":"2020-02-27","rows_on_this_dataset":7,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/nodis-neural-ordinary-differential-scene","title":"NODIS: Neural Ordinary Differential Scene Understanding","date":"2020-01-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-canonical-representations-for-scene","title":"Learning Canonical Representations for Scene Graph to Image Generation","date":"2019-12-16","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":2,"samples_unverified":15,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/image-synthesis-from-reconfigurable-layout","title":"Image Synthesis From Reconfigurable Layout and Style","date":"2019-08-20","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unsupervised-text-generation-from-structured","title":"An Unsupervised Joint System for Text Generation from Knowledge Graphs and Semantic Parsing","date":"2019-04-20","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/context-and-attribute-grounded-dense","title":"Context and Attribute Grounded Dense Captioning","date":"2019-04-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/knowledge-embedded-routing-network-for-scene","title":"Knowledge-Embedded Routing Network for Scene Graph Generation","date":"2019-03-08","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-to-compose-dynamic-tree-structures","title":"Learning to Compose Dynamic Tree Structures for Visual Contexts","date":"2018-12-05","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":4,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-level-multimodal-common-semantic-space","title":"Multi-level Multimodal Common Semantic Space for Image-Phrase Grounding","date":"2018-11-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/image-generation-from-layout","title":"Image Generation from Layout","date":"2018-11-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/graph-r-cnn-for-scene-graph-generation","title":"Graph R-CNN for Scene Graph Generation","date":"2018-08-01","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/image-generation-from-scene-graphs","title":"Image Generation from Scene Graphs","date":"2018-04-04","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":0,"samples_unverified":21,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scene-graph-generation-from-objects-phrases","title":"Scene Graph Generation from Objects, Phrases and Region Captions","date":"2017-07-31","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/modeling-relationships-in-referential","title":"Modeling Relationships in Referential Expressions with Compositional Modular Networks","date":"2016-11-30","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/densecap-fully-convolutional-localization","title":"DenseCap: Fully Convolutional Localization Networks for Dense Captioning","date":"2015-11-24","rows_on_this_dataset":2,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":16,"samples_harvested":142,"samples_ran":42,"samples_unverified":100,"pointer_only_for_licence":37,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}