{"url":"/task/visual-grounding","name":"Visual Grounding","slug":"visual-grounding","description_markdown":"Visual Grounding (VG) aims to locate the most relevant object or region in an image, based on a natural language query. The query can be a phrase, a sentence, or even a multi-round dialogue. There are three main challenges in\r\nVG: \r\n\r\n* What is the main focus in a query? \r\n* How to understand an image? \r\n* How to locate an object?","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":571,"papers_with_code":299,"benchmarks":4,"benchmark_tables_in_archive":4,"benchmark_tables_shown":4,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":11,"subtasks":3,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/visual-grounding-on-refcoco-testa","slug":"visual-grounding-on-refcoco-testa","dataset":"RefCOCO+ testA","dataset_url":"/dataset/refcoco","rows_in_archive":7,"metrics":["Accuracy (%)","IoU"],"first_row_in_archive_order":{"model":"Florence-2-large-ft","paper_title":"Florence-2: Advancing a Unified Representation for a Variety of Vision Tasks","paper_url":"/paper/florence-2-advancing-a-unified-representation","paper_date":"2023-11-10","arxiv_id":"2311.06242","code_links":[{"title":"retkowsky/florence-2","url":"https://github.com/retkowsky/florence-2"}],"syntology":null}},{"leaderboard":"/sota/visual-grounding-on-refcoco-test-b","slug":"visual-grounding-on-refcoco-test-b","dataset":"RefCOCO+ test B","dataset_url":"/dataset/refcoco","rows_in_archive":6,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"Florence-2-large-ft","paper_title":"Florence-2: Advancing a Unified Representation for a Variety of Vision Tasks","paper_url":"/paper/florence-2-advancing-a-unified-representation","paper_date":"2023-11-10","arxiv_id":"2311.06242","code_links":[{"title":"retkowsky/florence-2","url":"https://github.com/retkowsky/florence-2"}],"syntology":null}},{"leaderboard":"/sota/visual-grounding-on-refcoco-val","slug":"visual-grounding-on-refcoco-val","dataset":"RefCOCO+ val","dataset_url":"/dataset/refcoco","rows_in_archive":6,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"Florence-2-large-ft","paper_title":"Florence-2: Advancing a Unified Representation for a Variety of Vision Tasks","paper_url":"/paper/florence-2-advancing-a-unified-representation","paper_date":"2023-11-10","arxiv_id":"2311.06242","code_links":[{"title":"retkowsky/florence-2","url":"https://github.com/retkowsky/florence-2"}],"syntology":null}},{"leaderboard":"/sota/visual-grounding-on-refcoco-testa-1","slug":"visual-grounding-on-refcoco-testa-1","dataset":"RefCOCO testA","dataset_url":"/dataset/refcoco","rows_in_archive":1,"metrics":["IoU"],"first_row_in_archive_order":{"model":"HYDRA","paper_title":"HYDRA: A Hyper Agent for Dynamic Compositional Visual Reasoning","paper_url":"/paper/hydra-a-hyper-agent-for-dynamic-compositional","paper_date":"2024-03-19","arxiv_id":"2403.12884","code_links":[{"title":"ControlNet/HYDRA","url":"https://github.com/ControlNet/HYDRA"}],"syntology":{"n":8,"n_ran":5,"n_unverified":3,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/refcoco","name":"RefCOCO","full_name":"","num_papers_in_archive":439},{"url":"/dataset/rsvgd","name":"DIOR-RSVG","full_name":"","num_papers_in_archive":25},{"url":"/dataset/mrr-benchmark","name":"MRR-Benchmark","full_name":"Multi-Modal Reading Benchmark","num_papers_in_archive":11},{"url":"/dataset/infinity-mm","name":"Infinity-MM","full_name":"","num_papers_in_archive":5},{"url":"/dataset/skyeye-968k","name":"SkyEye-968k","full_name":"","num_papers_in_archive":5},{"url":"/dataset/a-game-of-sorts","name":"A Game Of Sorts","full_name":"","num_papers_in_archive":4},{"url":"/dataset/mono3drefer","name":"Mono3DRefer","full_name":"","num_papers_in_archive":3},{"url":"/dataset/vizwiz-answer-grounding","name":"VizWiz Answer Grounding","full_name":"","num_papers_in_archive":3},{"url":"/dataset/sk-vg","name":"SK-VG","full_name":"","num_papers_in_archive":2},{"url":"/dataset/automotiveui-bench-4k","name":"AutomotiveUI-Bench-4K","full_name":"","num_papers_in_archive":1},{"url":"/dataset/visargs","name":"VisArgs","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/3d-visual-grounding","name":"3D visual grounding"},{"url":"/task/person-centric-visual-grounding","name":"Person-centric Visual Grounding"},{"url":"/task/phrase-extraction-and-grounding-peg","name":"Phrase Extraction and Grounding (PEG)"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":299,"tagged_in_all":571,"items":[{"url":"/paper/vilbert-pretraining-task-agnostic","title":"ViLBERT: Pretraining Task-Agnostic Visiolinguistic Representations for Vision-and-Language Tasks","date":"2019-08-06","arxiv_id":"1908.02265","repositories_listed":11,"syntology":{"n":34,"n_ran":10,"n_unverified":24,"n_pointer_only":34}},{"url":"/paper/multimodal-compact-bilinear-pooling-for","title":"Multimodal Compact Bilinear Pooling for Visual Question Answering and Visual Grounding","date":"2016-06-06","arxiv_id":"1606.01847","repositories_listed":10,"syntology":null},{"url":"/paper/viewrefer-grasp-the-multi-view-knowledge-for","title":"ViewRefer: Grasp the Multi-view Knowledge for 3D Visual Grounding with GPT and Prototype Guidance","date":"2023-03-29","arxiv_id":"2303.16894","repositories_listed":7,"syntology":{"n":12,"n_ran":8,"n_unverified":4,"n_pointer_only":12}},{"url":"/paper/mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","arxiv_id":"2104.12763","repositories_listed":5,"syntology":{"n":11,"n_ran":6,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/towards-visual-grounding-a-survey","title":"Towards Visual Grounding: A Survey","date":"2024-12-28","arxiv_id":"2412.20206","repositories_listed":4,"syntology":{"n":15,"n_ran":9,"n_unverified":6,"n_pointer_only":3}},{"url":"/paper/set-of-mark-prompting-unleashes-extraordinary","title":"Set-of-Mark Prompting Unleashes Extraordinary Visual Grounding in GPT-4V","date":"2023-10-17","arxiv_id":"2310.11441","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/mplug-2-a-modularized-multi-modal-foundation","title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","date":"2023-02-01","arxiv_id":"2302.00402","repositories_listed":4,"syntology":{"n":19,"n_ran":9,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/unifying-architectures-tasks-and-modalities","title":"OFA: Unifying Architectures, Tasks, and Modalities Through a Simple Sequence-to-Sequence Learning Framework","date":"2022-02-07","arxiv_id":"2202.03052","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/orionbench-a-benchmark-for-chart-and-human","title":"OrionBench: A Benchmark for Chart and Human-Recognizable Object Detection in Infographics","date":"2025-05-23","arxiv_id":"2505.17473","repositories_listed":3,"syntology":null},{"url":"/paper/vrsbench-a-versatile-vision-language","title":"VRSBench: A Versatile Vision-Language Benchmark Dataset for Remote Sensing Image Understanding","date":"2024-06-18","arxiv_id":"2406.12384","repositories_listed":3,"syntology":{"n":13,"n_ran":11,"n_unverified":2,"n_pointer_only":13}},{"url":"/paper/docgenome-an-open-large-scale-scientific","title":"DocGenome: An Open Large-scale Scientific Document Benchmark for Training and Testing Multi-modal Large Language Models","date":"2024-06-17","arxiv_id":"2406.11633","repositories_listed":3,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/shapellm-universal-3d-object-understanding","title":"ShapeLLM: Universal 3D Object Understanding for Embodied Interaction","date":"2024-02-27","arxiv_id":"2402.17766","repositories_listed":3,"syntology":{"n":17,"n_ran":9,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/clip-vg-self-paced-curriculum-adapting-of","title":"CLIP-VG: Self-paced Curriculum Adapting of CLIP for Visual Grounding","date":"2023-05-15","arxiv_id":"2305.08685","repositories_listed":3,"syntology":null},{"url":"/paper/eda-explicit-text-decoupling-and-dense","title":"EDA: Explicit Text-Decoupling and Dense Alignment for 3D Visual Grounding","date":"2022-09-29","arxiv_id":"2209.14941","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/mplug-effective-and-efficient-vision-language","title":"mPLUG: Effective and Efficient Vision-Language Learning by Cross-modal Skip-connections","date":"2022-05-24","arxiv_id":"2205.12005","repositories_listed":3,"syntology":null},{"url":"/paper/collaborative-transformers-for-grounded","title":"Collaborative Transformers for Grounded Situation Recognition","date":"2022-03-30","arxiv_id":"2203.16518","repositories_listed":3,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/seqtr-a-simple-yet-universal-network-for","title":"SeqTR: A Simple yet Universal Network for Visual Grounding","date":"2022-03-30","arxiv_id":"2203.16265","repositories_listed":3,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/word-discovery-in-visually-grounded-self","title":"Word Discovery in Visually Grounded, Self-Supervised Speech Models","date":"2022-03-28","arxiv_id":"2203.15081","repositories_listed":3,"syntology":null},{"url":"/paper/jointly-learning-to-see-ask-and-guesswhat","title":"Beyond task success: A closer look at jointly learning to see, ask, and GuessWhat","date":"2018-09-10","arxiv_id":"1809.03408","repositories_listed":3,"syntology":null},{"url":"/paper/revisiting-visual-question-answering","title":"Revisiting Visual Question Answering Baselines","date":"2016-06-27","arxiv_id":"1606.08390","repositories_listed":3,"syntology":null},{"url":"/paper/grounding-of-textual-phrases-in-images-by","title":"Grounding of Textual Phrases in Images by Reconstruction","date":"2015-11-12","arxiv_id":"1511.03745","repositories_listed":3,"syntology":null},{"url":"/paper/enhancing-visual-grounding-for-gui-agents-via","title":"Enhancing Visual Grounding for GUI Agents via Self-Evolutionary Reinforcement Learning","date":"2025-05-18","arxiv_id":"2505.12370","repositories_listed":2,"syntology":null},{"url":"/paper/fixing-imbalanced-attention-to-mitigate-in","title":"PAINT: Paying Attention to INformed Tokens to Mitigate Hallucination in Large Vision-Language Model","date":"2025-01-21","arxiv_id":"2501.12206","repositories_listed":2,"syntology":null},{"url":"/paper/interpreting-object-level-foundation-models","title":"Interpreting Object-level Foundation Models via Visual Precision Search","date":"2024-11-25","arxiv_id":"2411.16198","repositories_listed":2,"syntology":null},{"url":"/paper/advancing-grounded-multimodal-named-entity","title":"Advancing Grounded Multimodal Named Entity Recognition via LLM-Based Reformulation and Box-Based Segmentation","date":"2024-06-11","arxiv_id":"2406.07268","repositories_listed":2,"syntology":null},{"url":"/paper/f-lmm-grounding-frozen-large-multimodal","title":"F-LMM: Grounding Frozen Large Multimodal Models","date":"2024-06-09","arxiv_id":"2406.05821","repositories_listed":2,"syntology":null},{"url":"/paper/h2rsvlm-towards-helpful-and-honest-remote","title":"VHM: Versatile and Honest Vision Language Model for Remote Sensing Image Analysis","date":"2024-03-29","arxiv_id":"2403.20213","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/lexicon-level-contrastive-visual-grounding","title":"Lexicon-Level Contrastive Visual-Grounding Improves Language Modeling","date":"2024-03-21","arxiv_id":"2403.14551","repositories_listed":2,"syntology":null},{"url":"/paper/seeing-is-believing-mitigating-hallucination","title":"Seeing is Believing: Mitigating Hallucination in Large Vision-Language Models via CLIP-Guided Decoding","date":"2024-02-23","arxiv_id":"2402.15300","repositories_listed":2,"syntology":{"n":17,"n_ran":11,"n_unverified":6,"n_pointer_only":17}},{"url":"/paper/llms-as-bridges-reformulating-grounded","title":"LLMs as Bridges: Reformulating Grounded Multimodal Named Entity Recognition","date":"2024-02-15","arxiv_id":"2402.09989","repositories_listed":2,"syntology":null}],"syntology_records":15,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}