{"url":"/task/phrase-grounding","name":"Phrase Grounding","slug":"phrase-grounding","description_markdown":"Given an image and a corresponding caption, the **Phrase Grounding** task aims to ground each entity mentioned by a noun phrase in the caption to a region in the image.\n\n\n<span class=\"description-source\">Source: [Phrase Grounding by Soft-Label Chain Conditional Random Field ](https://arxiv.org/abs/1909.00301)</span>","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":88,"papers_with_code":51,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":6,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/phrase-grounding-on-flickr30k-entities-test","slug":"phrase-grounding-on-flickr30k-entities-test","dataset":"Flickr30k Entities Test","dataset_url":"/dataset/flickr30k-entities","rows_in_archive":18,"metrics":["R@1","R@5","R@10"],"first_row_in_archive_order":{"model":"GLIPv2","paper_title":"GLIPv2: Unifying Localization and Vision-Language Understanding","paper_url":"/paper/glipv2-unifying-localization-and-vision","paper_date":"2022-06-12","arxiv_id":"2206.05836","code_links":[{"title":"microsoft/GLIP","url":"https://github.com/microsoft/GLIP"}],"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/phrase-grounding-on-flickr30k","slug":"phrase-grounding-on-flickr30k","dataset":"Flickr30k","dataset_url":"/dataset/flickr30k","rows_in_archive":3,"metrics":["Pointing Game Accuracy"],"first_row_in_archive_order":{"model":"GBS Ensemble + 12-in-1","paper_title":"Detector-Free Weakly Supervised Grounding by Separation","paper_url":"/paper/detector-free-weakly-supervised-grounding-by","paper_date":"2021-04-20","arxiv_id":"2104.09829","code_links":[{"title":"aarbelle/GroundingBySeparation","url":"https://github.com/aarbelle/GroundingBySeparation"}],"syntology":null}},{"leaderboard":"/sota/phrase-grounding-on-flickr30k-entities-dev","slug":"phrase-grounding-on-flickr30k-entities-dev","dataset":"Flickr30k Entities Dev","dataset_url":"/dataset/flickr30k-entities","rows_in_archive":3,"metrics":["R@1","R@10","R@5"],"first_row_in_archive_order":{"model":"Fiber-B","paper_title":"Coarse-to-Fine Vision-Language Pre-training with Fusion in the Backbone","paper_url":"/paper/coarse-to-fine-vision-language-pre-training","paper_date":"2022-06-15","arxiv_id":"2206.07643","code_links":[{"title":"microsoft/fiber","url":"https://github.com/microsoft/fiber"}],"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":2}}},{"leaderboard":"/sota/phrase-grounding-on-referit","slug":"phrase-grounding-on-referit","dataset":"ReferIt","dataset_url":null,"rows_in_archive":3,"metrics":["Pointing Game Accuracy","Accuracy"],"first_row_in_archive_order":{"model":"VG_BiLSTM_VGG","paper_title":"Multi-level Multimodal Common Semantic Space for Image-Phrase Grounding","paper_url":"/paper/multi-level-multimodal-common-semantic-space","paper_date":"2018-11-28","arxiv_id":"1811.11683","code_links":[{"title":"hassanhub/MultiGrounding","url":"https://github.com/hassanhub/MultiGrounding"}],"syntology":null}},{"leaderboard":"/sota/phrase-grounding-on-visual-genome","slug":"phrase-grounding-on-visual-genome","dataset":"Visual Genome","dataset_url":"/dataset/visual-genome","rows_in_archive":3,"metrics":["Pointing Game Accuracy"],"first_row_in_archive_order":{"model":"GbS VG","paper_title":"Detector-Free Weakly Supervised Grounding by Separation","paper_url":"/paper/detector-free-weakly-supervised-grounding-by","paper_date":"2021-04-20","arxiv_id":"2104.09829","code_links":[{"title":"aarbelle/GroundingBySeparation","url":"https://github.com/aarbelle/GroundingBySeparation"}],"syntology":null}}],"datasets":[{"url":"/dataset/visual-genome","name":"Visual Genome","full_name":"","num_papers_in_archive":1256},{"url":"/dataset/flickr30k","name":"Flickr30k","full_name":"Flickr30k","num_papers_in_archive":880},{"url":"/dataset/flickr30k-entities","name":"Flickr30K Entities","full_name":"","num_papers_in_archive":142},{"url":"/dataset/ms-cxr","name":"MS-CXR","full_name":"Making the Most of Text Semantics to Improve Biomedical Vision-Language Processing","num_papers_in_archive":32},{"url":"/dataset/g-vue","name":"G-VUE","full_name":"General-purpose Visual Understanding Evaluation","num_papers_in_archive":2},{"url":"/dataset/vd-ref","name":"VD-Ref","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/grounded-open-vocabulary-acquisition","name":"Grounded Open Vocabulary Acquisition"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":51,"tagged_in_all":88,"items":[{"url":"/paper/multimodal-compact-bilinear-pooling-for","title":"Multimodal Compact Bilinear Pooling for Visual Question Answering and Visual Grounding","date":"2016-06-06","arxiv_id":"1606.01847","repositories_listed":10,"syntology":null},{"url":"/paper/mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","arxiv_id":"2104.12763","repositories_listed":5,"syntology":{"n":11,"n_ran":6,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/towards-visual-grounding-a-survey","title":"Towards Visual Grounding: A Survey","date":"2024-12-28","arxiv_id":"2412.20206","repositories_listed":4,"syntology":{"n":15,"n_ran":9,"n_unverified":6,"n_pointer_only":3}},{"url":"/paper/open-vocabulary-phrase-detection","title":"Revisiting Image-Language Networks for Open-ended Phrase Detection","date":"2018-11-17","arxiv_id":"1811.07212","repositories_listed":3,"syntology":null},{"url":"/paper/grounding-of-textual-phrases-in-images-by","title":"Grounding of Textual Phrases in Images by Reconstruction","date":"2015-11-12","arxiv_id":"1511.03745","repositories_listed":3,"syntology":null},{"url":"/paper/an-open-and-comprehensive-pipeline-for","title":"An Open and Comprehensive Pipeline for Unified Object Grounding and Detection","date":"2024-01-04","arxiv_id":"2401.02361","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/kosmos-2-grounding-multimodal-large-language","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","date":"2023-06-26","arxiv_id":"2306.14824","repositories_listed":2,"syntology":null},{"url":"/paper/making-the-most-of-text-semantics-to-improve","title":"Making the Most of Text Semantics to Improve Biomedical Vision--Language Processing","date":"2022-04-21","arxiv_id":"2204.09817","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/flickr30k-entities-collecting-region-to","title":"Flickr30k Entities: Collecting Region-to-Phrase Correspondences for Richer Image-to-Sentence Models","date":"2015-05-19","arxiv_id":"1505.04870","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/disambiguating-reference-in-visually-grounded","title":"Disambiguating Reference in Visually Grounded Dialogues through Joint Modeling of Textual and Multimodal Semantic Structures","date":"2025-05-16","arxiv_id":"2505.11726","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/anatomical-grounding-pre-training-for-medical","title":"Anatomical grounding pre-training for medical phrase grounding","date":"2025-02-23","arxiv_id":"2502.16585","repositories_listed":1,"syntology":null},{"url":"/paper/vicca-visual-interpretation-and-comprehension","title":"VICCA: Visual Interpretation and Comprehension of Chest X-ray Anomalies in Generated Report Without Human Feedback","date":"2025-01-29","arxiv_id":"2501.17726","repositories_listed":1,"syntology":null},{"url":"/paper/context-infused-visual-grounding-for-art","title":"Context-Infused Visual Grounding for Art","date":"2024-10-16","arxiv_id":"2410.12369","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/transformer-with-controlled-attention-for","title":"Transformer with Controlled Attention for Synchronous Motion Captioning","date":"2024-09-13","arxiv_id":"2409.09177","repositories_listed":1,"syntology":null},{"url":"/paper/lightmdetr-a-lightweight-approach-for-low","title":"A Lightweight Modular Framework for Low-Cost Open-Vocabulary Object Detection Training","date":"2024-08-20","arxiv_id":"2408.10787","repositories_listed":1,"syntology":null},{"url":"/paper/empathic-grounding-explorations-using","title":"Empathic Grounding: Explorations using Multimodal Interaction and Large Language Models with Conversational Agents","date":"2024-07-01","arxiv_id":"2407.01824","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-medical-phrase-grounding-with-off","title":"Zero-Shot Medical Phrase Grounding with Off-the-shelf Diffusion Models","date":"2024-04-19","arxiv_id":"2404.12920","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/griffon-v2-advancing-multimodal-perception","title":"Griffon v2: Advancing Multimodal Perception with High-Resolution Scaling and Visual-Language Co-Referring","date":"2024-03-14","arxiv_id":"2403.09333","repositories_listed":1,"syntology":null},{"url":"/paper/pg-video-llava-pixel-grounding-large-video","title":"PG-Video-LLaVA: Pixel Grounding Large Video-Language Models","date":"2023-11-22","arxiv_id":"2311.13435","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":5}},{"url":"/paper/augment-the-pairs-semantics-preserving-image","title":"Augment the Pairs: Semantics-Preserving Image-Caption Pair Augmentation for Grounding-Based Vision and Language Models","date":"2023-11-05","arxiv_id":"2311.02536","repositories_listed":1,"syntology":null},{"url":"/paper/localizing-active-objects-from-egocentric","title":"Localizing Active Objects from Egocentric Vision with Symbolic World Knowledge","date":"2023-10-23","arxiv_id":"2310.15066","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-representation-in-radiography","title":"Enhancing Representation in Radiography-Reports Foundation Model: A Granular Alignment Algorithm Using Masked Contrastive Learning","date":"2023-09-12","arxiv_id":"2309.05904","repositories_listed":1,"syntology":null},{"url":"/paper/box-based-refinement-for-weakly-supervised","title":"Box-based Refinement for Weakly Supervised and Unsupervised Localization Tasks","date":"2023-09-07","arxiv_id":"2309.03874","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_unverified":5,"n_pointer_only":11}},{"url":"/paper/a-joint-study-of-phrase-grounding-and-task","title":"A Joint Study of Phrase Grounding and Task Performance in Vision and Language Models","date":"2023-09-06","arxiv_id":"2309.02691","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_unverified":2,"n_pointer_only":12}},{"url":"/paper/a-survey-on-interpretable-cross-modal","title":"A Survey on Interpretable Cross-modal Reasoning","date":"2023-09-05","arxiv_id":"2309.01955","repositories_listed":1,"syntology":null},{"url":"/paper/pay-attention-accuracy-versus","title":"Trade-offs in Fine-tuned Diffusion Models Between Accuracy and Interpretability","date":"2023-03-31","arxiv_id":"2303.17908","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-exploit-temporal-structure-for","title":"Learning to Exploit Temporal Structure for Biomedical Vision-Language Processing","date":"2023-01-11","arxiv_id":"2301.04558","repositories_listed":1,"syntology":null},{"url":"/paper/similarity-maps-for-self-training-weakly","title":"Similarity Maps for Self-Training Weakly-Supervised Phrase Grounding","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dq-detr-dual-query-detection-transformer-for","title":"DQ-DETR: Dual Query Detection Transformer for Phrase Extraction and Grounding","date":"2022-11-28","arxiv_id":"2211.15516","repositories_listed":1,"syntology":null},{"url":"/paper/extending-phrase-grounding-with-pronouns-in","title":"Extending Phrase Grounding with Pronouns in Visual Dialogues","date":"2022-10-23","arxiv_id":"2210.12658","repositories_listed":1,"syntology":{"n":9,"n_ran":0,"n_unverified":9,"n_pointer_only":0}}],"syntology_records":12,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}