{"url":"/task/grounded-situation-recognition","name":"Grounded Situation Recognition","slug":"grounded-situation-recognition","description_markdown":"Grounded Situation Recognition aims to produce the structured image summary which describes the primary activity (verb), its relevant entities (nouns), and their bounding-box groundings.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":15,"papers_with_code":11,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/grounded-situation-recognition-on-swig","slug":"grounded-situation-recognition-on-swig","dataset":"SWiG","dataset_url":null,"rows_in_archive":13,"metrics":["Top-1 Verb","Top-1 Verb & Grounded-Value","Top-1 Verb & Value","Top-5 Verbs","Top-5 Verbs & Grounded-Value","Top-5 Verbs & Value"],"first_row_in_archive_order":{"model":"Ours (CoFormer+)","paper_title":"Dynamic Scene Understanding from Vision-Language Representations","paper_url":"/paper/dynamic-scene-understanding-from-vision","paper_date":"2025-01-20","arxiv_id":"2501.11653","code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/vasr","name":"VASR","full_name":"Visual Analogies of Situation Recognition","num_papers_in_archive":4}],"subtasks":[],"parent_tasks":[{"url":"/task/situation-recognition","name":"Situation Recognition"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":11,"of":11,"tagged_in_all":15,"items":[{"url":"/paper/collaborative-transformers-for-grounded","title":"Collaborative Transformers for Grounded Situation Recognition","date":"2022-03-30","arxiv_id":"2203.16518","repositories_listed":3,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/commonly-uncommon-semantic-sparsity-in","title":"Commonly Uncommon: Semantic Sparsity in Situation Recognition","date":"2016-12-03","arxiv_id":"1612.00901","repositories_listed":2,"syntology":null},{"url":"/paper/open-scene-understanding-grounded-situation","title":"Open Scene Understanding: Grounded Situation Recognition Meets Segment Anything for Helping People with Visual Impairments","date":"2023-07-15","arxiv_id":"2307.07757","repositories_listed":1,"syntology":null},{"url":"/paper/clipsitu-effectively-leveraging-clip-for","title":"ClipSitu: Effectively Leveraging CLIP for Conditional Predictions in Situation Recognition","date":"2023-07-02","arxiv_id":"2307.00586","repositories_listed":1,"syntology":null},{"url":"/paper/gsrformer-grounded-situation-recognition","title":"GSRFormer: Grounded Situation Recognition Transformer with Alternate Semantic Attention Refinement","date":"2022-08-18","arxiv_id":"2208.08965","repositories_listed":1,"syntology":{"n":11,"n_ran":4,"n_unverified":7,"n_pointer_only":11}},{"url":"/paper/rethinking-the-two-stage-framework-for","title":"Rethinking the Two-Stage Framework for Grounded Situation Recognition","date":"2021-12-10","arxiv_id":"2112.05375","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_unverified":3,"n_pointer_only":12}},{"url":"/paper/grounded-situation-recognition-with","title":"Grounded Situation Recognition with Transformers","date":"2021-11-19","arxiv_id":"2111.10135","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_unverified":5,"n_pointer_only":9}},{"url":"/paper/attention-based-context-aware-reasoning-for","title":"Attention-Based Context Aware Reasoning for Situation Recognition","date":"2020-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/grounded-situation-recognition","title":"Grounded Situation Recognition","date":"2020-03-26","arxiv_id":"2003.12058","repositories_listed":1,"syntology":null},{"url":"/paper/situation-recognition-with-graph-neural","title":"Situation Recognition with Graph Neural Networks","date":"2017-08-14","arxiv_id":"1708.04320","repositories_listed":1,"syntology":null},{"url":"/paper/situation-recognition-visual-semantic-role","title":"Situation Recognition: Visual Semantic Role Labeling for Image Understanding","date":"2016-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null}],"syntology_records":4,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}