{"url":"/task/scene-graph-generation","name":"Scene Graph Generation","slug":"scene-graph-generation","description_markdown":"A scene graph is a structured representation of an image, where nodes in a scene graph correspond to object bounding boxes with their object categories, and edges correspond to their pairwise relationships between objects. The task of **Scene Graph Generation** is to generate a visually-grounded scene graph that most accurately correlates with an image.\n\n\n<span class=\"description-source\">Source: [Scene Graph Generation by Iterative Message Passing ](https://arxiv.org/abs/1701.02426)</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":318,"papers_with_code":151,"benchmarks":7,"benchmark_tables_in_archive":7,"benchmark_tables_shown":7,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":11,"subtasks":2,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/scene-graph-generation-on-visual-genome","slug":"scene-graph-generation-on-visual-genome","dataset":"Visual Genome","dataset_url":"/dataset/visual-genome","rows_in_archive":19,"metrics":["Recall@50","mean Recall @20","Recall@100","Recall@20","mean Recall @100","R@100","mR@100","mR@50","zR@100","zR@20","zR@50"],"first_row_in_archive_order":{"model":"SpeaQ (without reweighting)","paper_title":"Groupwise Query Specialization and Quality-Aware Multi-Assignment for Transformer-based Visual Relationship Detection","paper_url":"/paper/groupwise-query-specialization-and-quality","paper_date":"2024-03-26","arxiv_id":"2403.17709","code_links":[{"title":"mlvlab/speaq","url":"https://github.com/mlvlab/speaq"}],"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":3}}},{"leaderboard":"/sota/scene-graph-generation-on-4d-or","slug":"scene-graph-generation-on-4d-or","dataset":"4D-OR","dataset_url":"/dataset/4d-or","rows_in_archive":5,"metrics":["F1"],"first_row_in_archive_order":{"model":"ORacle","paper_title":"ORacle: Large Vision-Language Models for Knowledge-Guided Holistic OR Domain Modeling","paper_url":"/paper/oracle-large-vision-language-models-for","paper_date":"2024-04-10","arxiv_id":"2404.07031","code_links":[{"title":"egeozsoy/Oracle","url":"https://github.com/egeozsoy/Oracle"}],"syntology":null}},{"leaderboard":"/sota/scene-graph-generation-on-3r-scan-1","slug":"scene-graph-generation-on-3r-scan-1","dataset":"3R-Scan","dataset_url":"/dataset/3r-scan","rows_in_archive":2,"metrics":["Top-5 Accuracy"],"first_row_in_archive_order":{"model":"SceneGraphFusion","paper_title":"SceneGraphFusion: Incremental 3D Scene Graph Prediction from RGB-D Sequences","paper_url":"/paper/scenegraphfusion-incremental-3d-scene-graph","paper_date":"2021-03-27","arxiv_id":"2103.14898","code_links":[{"title":"ShunChengWu/SceneGraphFusion","url":"https://github.com/ShunChengWu/SceneGraphFusion"},{"title":"ShunChengWu/3DSSG","url":"https://github.com/ShunChengWu/3DSSG"}],"syntology":null}},{"leaderboard":"/sota/scene-graph-generation-on-vrd","slug":"scene-graph-generation-on-vrd","dataset":"VRD","dataset_url":"/dataset/vrd","rows_in_archive":2,"metrics":["Recall@50"],"first_row_in_archive_order":{"model":"FactorizableNet","paper_title":"Factorizable Net: An Efficient Subgraph-based Framework for Scene Graph Generation","paper_url":"/paper/factorizable-net-an-efficient-subgraph-based","paper_date":"2018-06-29","arxiv_id":"1806.11538","code_links":[{"title":"yikang-li/FactorizableNet","url":"https://github.com/yikang-li/FactorizableNet"}],"syntology":null}},{"leaderboard":"/sota/scene-graph-generation-on-gqa","slug":"scene-graph-generation-on-gqa","dataset":"GQA","dataset_url":"/dataset/gqa","rows_in_archive":1,"metrics":["zR@100","zR@20","zR@50"],"first_row_in_archive_order":{"model":"KnowZRel","paper_title":"KnowZRel: Common Sense Knowledge-based Zero-Shot Relationship Retrieval for Generalised Scene Graph Generation","paper_url":"/paper/knowzrel-common-sense-knowledge-based-zero","paper_date":"2025-02-21","arxiv_id":null,"code_links":[{"title":"jaleedkhan/zsrr-sgg","url":"https://github.com/jaleedkhan/zsrr-sgg"}],"syntology":null}},{"leaderboard":"/sota/scene-graph-generation-on-mm-or","slug":"scene-graph-generation-on-mm-or","dataset":"MM-OR","dataset_url":"/dataset/mm-or","rows_in_archive":1,"metrics":["Macro F1"],"first_row_in_archive_order":{"model":"MM2SG","paper_title":"MM-OR: A Large Multimodal Operating Room Dataset for Semantic Understanding of High-Intensity Surgical Environments","paper_url":"/paper/mm-or-a-large-multimodal-operating-room","paper_date":"2025-03-04","arxiv_id":"2503.02579","code_links":[{"title":"egeozsoy/MM-OR","url":"https://github.com/egeozsoy/MM-OR"}],"syntology":null}},{"leaderboard":"/sota/scene-graph-generation-on-ms-coco","slug":"scene-graph-generation-on-ms-coco","dataset":"MS-COCO","dataset_url":"/dataset/coco","rows_in_archive":1,"metrics":["R@100","R@20","R@50","mR@100","mR@20","mR@50"],"first_row_in_archive_order":{"model":"NeuSyRE","paper_title":"NeuSyRE: Neuro-Symbolic Visual Understanding and Reasoning Framework based on Scene Graph Enrichment","paper_url":"/paper/neusyre-neuro-symbolic-visual-understanding","paper_date":"2023-11-05","arxiv_id":null,"code_links":[{"title":"jaleedkhan/neusire","url":"https://github.com/jaleedkhan/neusire"}],"syntology":null}}],"datasets":[{"url":"/dataset/coco","name":"COCO (Common Objects in Context)","full_name":"Common Objects in Context","num_papers_in_archive":11922},{"url":"/dataset/visual-genome","name":"Visual Genome","full_name":"","num_papers_in_archive":1256},{"url":"/dataset/gqa","name":"GQA","full_name":"GQA","num_papers_in_archive":749},{"url":"/dataset/vrd","name":"VRD","full_name":"Visual Relationship Detection dataset","num_papers_in_archive":151},{"url":"/dataset/3r-scan","name":"3RScan","full_name":"","num_papers_in_archive":44},{"url":"/dataset/3dssg","name":"3DSSG","full_name":"","num_papers_in_archive":36},{"url":"/dataset/psg-dataset","name":"PSG Dataset","full_name":"","num_papers_in_archive":28},{"url":"/dataset/4d-or","name":"4D-OR","full_name":"","num_papers_in_archive":11},{"url":"/dataset/mm-or","name":"MM-OR","full_name":"","num_papers_in_archive":3},{"url":"/dataset/haystack","name":"Haystack","full_name":"","num_papers_in_archive":1},{"url":"/dataset/spacesgg","name":"SpaceSGG","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/panoptic-scene-graph-generation","name":"Panoptic Scene Graph Generation"},{"url":"/task/unbiased-scene-graph-generation","name":"Unbiased Scene Graph Generation"}],"parent_tasks":[{"url":"/task/scene-parsing","name":"Scene Parsing"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":151,"tagged_in_all":318,"items":[{"url":"/paper/unbiased-scene-graph-generation-from-biased","title":"Unbiased Scene Graph Generation from Biased Training","date":"2020-02-27","arxiv_id":"2002.11949","repositories_listed":6,"syntology":{"n":6,"n_ran":3,"n_unverified":3,"n_pointer_only":6}},{"url":"/paper/learning-to-compose-dynamic-tree-structures","title":"Learning to Compose Dynamic Tree Structures for Visual Contexts","date":"2018-12-05","arxiv_id":"1812.01880","repositories_listed":6,"syntology":{"n":14,"n_ran":4,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/scene-graph-generation-by-iterative-message","title":"Scene Graph Generation by Iterative Message Passing","date":"2017-01-10","arxiv_id":"1701.02426","repositories_listed":5,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/structured-sparse-r-cnn-for-direct-scene","title":"Structured Sparse R-CNN for Direct Scene Graph Generation","date":"2021-06-21","arxiv_id":"2106.10815","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/bipartite-graph-network-with-adaptive-message","title":"Bipartite Graph Network with Adaptive Message Passing for Unbiased Scene Graph Generation","date":"2021-04-01","arxiv_id":"2104.00308","repositories_listed":4,"syntology":null},{"url":"/paper/scene-graph-generation-in-large-size-vhr","title":"STAR: A First-Ever Dataset and A Large-Scale Benchmark for Scene Graph Generation in Large-Size Satellite Imagery","date":"2024-06-13","arxiv_id":"2406.09410","repositories_listed":3,"syntology":null},{"url":"/paper/4d-panoptic-scene-graph-generation-1","title":"4D Panoptic Scene Graph Generation","date":"2024-05-16","arxiv_id":"2405.10305","repositories_listed":3,"syntology":{"n":14,"n_ran":13,"n_unverified":1,"n_pointer_only":10}},{"url":"/paper/panoptic-video-scene-graph-generation-1","title":"Panoptic Video Scene Graph Generation","date":"2023-11-28","arxiv_id":"2311.17058","repositories_listed":3,"syntology":{"n":11,"n_ran":11,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/rlipv2-fast-scaling-of-relational-language","title":"RLIPv2: Fast Scaling of Relational Language-Image Pre-training","date":"2023-08-18","arxiv_id":"2308.09351","repositories_listed":3,"syntology":{"n":30,"n_ran":22,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/sgdraw-scene-graph-drawing-interface-using","title":"SGDraw: Scene Graph Drawing Interface Using Object-Oriented Representation","date":"2022-11-30","arxiv_id":"2211.16697","repositories_listed":3,"syntology":null},{"url":"/paper/knowledge-embedded-routing-network-for-scene","title":"Knowledge-Embedded Routing Network for Scene Graph Generation","date":"2019-03-08","arxiv_id":"1903.03326","repositories_listed":3,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/graphical-contrastive-losses-for-scene-graph","title":"Graphical Contrastive Losses for Scene Graph Parsing","date":"2019-03-07","arxiv_id":"1903.02728","repositories_listed":3,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/linknet-relational-embedding-for-scene-graph","title":"LinkNet: Relational Embedding for Scene Graph","date":"2018-11-15","arxiv_id":"1811.06410","repositories_listed":3,"syntology":null},{"url":"/paper/graph-r-cnn-for-scene-graph-generation","title":"Graph R-CNN for Scene Graph Generation","date":"2018-08-01","arxiv_id":"1808.00191","repositories_listed":3,"syntology":null},{"url":"/paper/pixels-to-graphs-by-associative-embedding","title":"Pixels to Graphs by Associative Embedding","date":"2017-06-22","arxiv_id":"1706.07365","repositories_listed":3,"syntology":null},{"url":"/paper/manga109dialog-a-large-scale-dialogue-dataset","title":"Manga109Dialog: A Large-scale Dialogue Dataset for Comics Speaker Detection","date":"2023-06-30","arxiv_id":"2306.17469","repositories_listed":2,"syntology":null},{"url":"/paper/fine-grained-scene-graph-generation-with-data","title":"Fine-Grained Scene Graph Generation with Data Transfer","date":"2022-03-22","arxiv_id":"2203.11654","repositories_listed":2,"syntology":null},{"url":"/paper/spatial-temporal-transformer-for-dynamic","title":"Spatial-Temporal Transformer for Dynamic Scene Graph Generation","date":"2021-07-26","arxiv_id":"2107.12309","repositories_listed":2,"syntology":{"n":14,"n_ran":7,"n_unverified":7,"n_pointer_only":5}},{"url":"/paper/scenegraphfusion-incremental-3d-scene-graph","title":"SceneGraphFusion: Incremental 3D Scene Graph Prediction from RGB-D Sequences","date":"2021-03-27","arxiv_id":"2103.14898","repositories_listed":2,"syntology":null},{"url":"/paper/learning-and-reasoning-with-the-graph","title":"Learning and Reasoning with the Graph Structure Representation in Robotic Surgery","date":"2020-07-07","arxiv_id":"2007.03357","repositories_listed":2,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":8}},{"url":"/paper/learning-visual-commonsense-for-robust-scene","title":"Learning Visual Commonsense for Robust Scene Graph Generation","date":"2020-06-17","arxiv_id":"2006.09623","repositories_listed":2,"syntology":{"n":11,"n_ran":3,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/visual-graphs-from-motion-vgfm-scene","title":"Visual Graphs from Motion (VGfM): Scene understanding with object geometry reasoning","date":"2018-07-16","arxiv_id":"1807.05933","repositories_listed":2,"syntology":null},{"url":"/paper/open-world-scene-graph-generation-using","title":"Open World Scene Graph Generation using Vision Language Models","date":"2025-06-09","arxiv_id":"2506.08189","repositories_listed":1,"syntology":null},{"url":"/paper/egoexor-an-ego-exo-centric-operating-room","title":"EgoExOR: An Ego-Exo-Centric Operating Room Dataset for Surgical Activity Understanding","date":"2025-05-30","arxiv_id":"2505.24287","repositories_listed":1,"syntology":null},{"url":"/paper/llm-meets-scene-graph-can-large-language","title":"LLM Meets Scene Graph: Can Large Language Models Understand and Generate Scene Graphs? A Benchmark and Empirical Study","date":"2025-05-26","arxiv_id":"2505.19510","repositories_listed":1,"syntology":null},{"url":"/paper/scenir-visual-semantic-clarity-through","title":"SCENIR: Visual Semantic Clarity through Unsupervised Scene Graph Retrieval","date":"2025-05-21","arxiv_id":"2505.15867","repositories_listed":1,"syntology":null},{"url":"/paper/diffvsgg-diffusion-driven-online-video-scene","title":"DIFFVSGG: Diffusion-Driven Online Video Scene Graph Generation","date":"2025-03-18","arxiv_id":"2503.13957","repositories_listed":1,"syntology":null},{"url":"/paper/mm-or-a-large-multimodal-operating-room","title":"MM-OR: A Large Multimodal Operating Room Dataset for Semantic Understanding of High-Intensity Surgical Environments","date":"2025-03-04","arxiv_id":"2503.02579","repositories_listed":1,"syntology":null},{"url":"/paper/knowzrel-common-sense-knowledge-based-zero","title":"KnowZRel: Common Sense Knowledge-based Zero-Shot Relationship Retrieval for Generalised Scene Graph Generation","date":"2025-02-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-video-scene-graph","title":"Weakly Supervised Video Scene Graph Generation via Natural Language Supervision","date":"2025-02-21","arxiv_id":"2502.15370","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}}],"syntology_records":13,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}