{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-grounding/papers/6","list_of":"/task/visual-grounding","task":"Visual Grounding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":6,"rows_per_page":100,"rows":[501,571],"of":571,"counts":{"archive_papers_tagged":571,"with_a_code_link":299,"where_syntology_ran_a_sample":111,"not_listed_spam_title":0,"listed":571,"listed_where_code_ran":111,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":95,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":95,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-grounding","prev":"/task/visual-grounding/papers/5","next":null,"papers":[{"url":null,"slug":"suspected-object-matters-rethinking-model-s","title":"Suspected Object Matters: Rethinking Model's Prediction for One-stage Visual Grounding","date":"2022-03-10","arxiv_id":"2203.05186","repositories_listed":0,"syntology":null},{"url":"/paper/3djcg-a-unified-framework-for-joint-dense","slug":"3djcg-a-unified-framework-for-joint-dense","title":"3DJCG: A Unified Framework for Joint Dense Captioning and Visual Grounding on 3D Point Clouds","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deconfounded-visual-grounding","title":"Deconfounded Visual Grounding","date":"2021-12-31","arxiv_id":"2112.15324","repositories_listed":0,"syntology":null},{"url":null,"slug":"rovist-learning-robust-metrics-for-visual","title":"RoViST: Learning Robust Metrics for Visual Storytelling","date":"2021-12-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/d3net-a-speaker-listener-architecture-for","slug":"d3net-a-speaker-listener-architecture-for","title":"D3Net: A Unified Speaker-Listener Architecture for 3D Dense Captioning and Visual Grounding","date":"2021-12-02","arxiv_id":"2112.01551","repositories_listed":0,"syntology":null},{"url":null,"slug":"less-is-more-generating-grounded-navigation","title":"Less is More: Generating Grounded Navigation Instructions from Landmarks","date":"2021-11-25","arxiv_id":"2111.12872","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-visual-grounding-of-referring","title":"Zero-Shot Visual Grounding of Referring Utterances in Dialogue","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-as-grounding-exploring-textual-and","title":"Attention as Grounding: Exploring Textual and Cross-Modal Attention on Entities and Relations in Language-and-Vision Transformer","date":"2021-10-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-multi-modal-embeddings-from","title":"Efficient Multi-Modal Embeddings from Structured Data","date":"2021-10-06","arxiv_id":"2110.02577","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieve-caption-generate-visual-grounding","title":"Retrieve, Caption, Generate: Visual Grounding for Enhancing Commonsense in Text Generation Models","date":"2021-09-08","arxiv_id":"2109.03892","repositories_listed":0,"syntology":null},{"url":null,"slug":"invigorate-interactive-visual-grounding-and","title":"INVIGORATE: Interactive Visual Grounding and Grasping in Clutter","date":"2021-08-25","arxiv_id":"2108.11092","repositories_listed":0,"syntology":null},{"url":null,"slug":"transrefer3d-entity-and-relation-aware","title":"TransRefer3D: Entity-and-Relation Aware Transformer for Fine-Grained 3D Visual Grounding","date":"2021-08-05","arxiv_id":"2108.02388","repositories_listed":0,"syntology":null},{"url":null,"slug":"attending-self-attention-a-case-study-of","title":"Attending Self-Attention: A Case Study of Visually Grounded Supervision in Vision-and-Language Transformers","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"word2pix-word-to-pixel-cross-attention","title":"Word2Pix: Word to Pixel Cross Attention Transformer in Visual Grounding","date":"2021-07-31","arxiv_id":"2108.00205","repositories_listed":0,"syntology":null},{"url":null,"slug":"languagerefer-spatial-language-model-for-3d","title":"LanguageRefer: Spatial-Language Model for 3D Visual Grounding","date":"2021-07-07","arxiv_id":"2107.03438","repositories_listed":0,"syntology":null},{"url":null,"slug":"adventurer-s-treasure-hunt-a-transparent","title":"Adventurer's Treasure Hunt: A Transparent System for Visually Grounded Compositional Visual Question Answering based on Scene Graphs","date":"2021-06-28","arxiv_id":"2106.14476","repositories_listed":0,"syntology":null},{"url":null,"slug":"aifit-automatic-3d-human-interpretable","title":"AIFit: Automatic 3D Human-Interpretable Feedback Models for Fitness Training","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-keyword-localisation-in","title":"Attention-Based Keyword Localisation in Speech using Visual Grounding","date":"2021-06-16","arxiv_id":"2106.08859","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-grounding-strategies-for-text-only","title":"Visual Grounding Strategies for Text-Only Natural Language Processing","date":"2021-03-25","arxiv_id":"2103.13942","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-intuitive-agent-for-remote-embodied","title":"Scene-Intuitive Agent for Remote Embodied Visual Grounding","date":"2021-03-24","arxiv_id":"2103.12944","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoupled-spatial-temporal-graphs-for-generic","title":"Decoupled Spatial Temporal Graphs for Generic Visual Grounding","date":"2021-03-18","arxiv_id":"2103.10191","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-visual-grounding-for-natural-human","title":"Few-Shot Visual Grounding for Natural Human-Robot Interaction","date":"2021-03-17","arxiv_id":"2103.09720","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-in-vision-a-survey","title":"Transformers in Vision: A Survey","date":"2021-01-04","arxiv_id":"2101.01169","repositories_listed":0,"syntology":null},{"url":null,"slug":"3dvg-transformer-relation-modeling-for-visual","title":"3DVG-Transformer: Relation Modeling for Visual Grounding on Point Clouds","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-video-entailment-with-grounded","title":"Explainable Video Entailment With Grounded Visual Evidence","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"casting-your-model-learning-to-localize","title":"CASTing Your Model: Learning to Localize Improves Self-Supervised Representations","date":"2020-12-08","arxiv_id":"2012.04630","repositories_listed":0,"syntology":null},{"url":"/paper/class-agnostic-object-detection","slug":"class-agnostic-object-detection","title":"Class-agnostic Object Detection","date":"2020-11-28","arxiv_id":"2011.14204","repositories_listed":0,"syntology":null},{"url":null,"slug":"commands-4-autonomous-vehicles-c4av-workshop","title":"Commands 4 Autonomous Vehicles (C4AV) Workshop Summary","date":"2020-09-18","arxiv_id":"2009.08792","repositories_listed":0,"syntology":null},{"url":null,"slug":"propagating-over-phrase-relations-for-one","title":"Propagating Over Phrase Relations for One-Stage Visual Grounding","date":"2020-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-language-biases-in-visual-question","title":"Reducing Language Biases in Visual Question Answering with Visually-Grounded Question Encoder","date":"2020-07-13","arxiv_id":"2007.06198","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-granularity-modularized-network-for","title":"Multi-Granularity Modularized Network for Abstract Visual Reasoning","date":"2020-07-09","arxiv_id":"2007.04670","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-supports-visual-language-grounding","title":"Knowledge Supports Visual Language Grounding: A Case Study on Colour Terms","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-visual-grounding-in-interaction-bringing","title":"Fast visual grounding in interaction: bringing few-shot learning with neural networks to an interactive robot","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-grounding-annotation-of-recipe-flow","title":"Visual Grounding Annotation of Recipe Flow Graph","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"spatio-temporal-graph-for-video-captioning","title":"Spatio-Temporal Graph for Video Captioning with Knowledge Distillation","date":"2020-03-31","arxiv_id":"2003.13942","repositories_listed":0,"syntology":null},{"url":null,"slug":"giving-commands-to-a-self-driving-car-a","title":"Giving Commands to a Self-driving Car: A Multimodal Reasoner for Visual Grounding","date":"2020-03-19","arxiv_id":"2003.08717","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-communication-with-world-models","title":"Emergent Communication with World Models","date":"2020-02-22","arxiv_id":"2002.09604","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-context-attention-and-audio","title":"Exploring Context, Attention and Audio Features for Audio Visual Scene-Aware Dialog","date":"2019-12-20","arxiv_id":"1912.10132","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-temporal-visual-grounding-of","title":"Compositional Temporal Visual Grounding of Natural Language Event Descriptions","date":"2019-12-04","arxiv_id":"1912.02256","repositories_listed":0,"syntology":null},{"url":null,"slug":"optibox-breaking-the-limits-of-proposals-for","title":"OptiBox: Breaking the Limits of Proposals for Visual Grounding","date":"2019-11-29","arxiv_id":"1912.00076","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-past-references-for-robust","title":"Leveraging Past References for Robust Language Grounding","date":"2019-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"countering-language-drift-via-visual","title":"Countering Language Drift via Visual Grounding","date":"2019-09-10","arxiv_id":"1909.04499","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-disentanglement-filter-an-1","title":"Differentiable Disentanglement Filter: an Application Agnostic Core Concept Discovery Probe","date":"2019-09-04","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-unified-attention-networks-for","title":"Multimodal Unified Attention Networks for Vision-and-Language Interactions","date":"2019-08-12","arxiv_id":"1908.04107","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-disentanglement-filter-an","title":"Differentiable Disentanglement Filter: an Application Agnostic Core Concept Discovery Probe","date":"2019-07-17","arxiv_id":"1907.07507","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-from-audio-visual-grounding","title":"Transfer Learning from Audio-Visual Grounding to Speech Recognition","date":"2019-07-09","arxiv_id":"1907.04355","repositories_listed":0,"syntology":null},{"url":null,"slug":"referring-expression-grounding-by","title":"Joint Visual Grounding with Language Scene Graphs","date":"2019-06-09","arxiv_id":"1906.03561","repositories_listed":0,"syntology":null},{"url":null,"slug":"visually-grounded-neural-syntax-acquisition","title":"Visually Grounded Neural Syntax Acquisition","date":"2019-06-07","arxiv_id":"1906.02890","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-compose-and-reason-with-language","title":"Learning to Compose and Reason with Language Tree Structures for Visual Grounding","date":"2019-06-05","arxiv_id":"1906.01784","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-contributions-of-visual-and-textual","title":"On the Contributions of Visual and Textual Supervision in Low-Resource Semantic Speech Retrieval","date":"2019-04-24","arxiv_id":"1904.10947","repositories_listed":0,"syntology":null},{"url":"/paper/vqd-visual-query-detection-in-natural-scenes","slug":"vqd-visual-query-detection-in-natural-scenes","title":"VQD: Visual Query Detection in Natural Scenes","date":"2019-04-04","arxiv_id":"1904.02794","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-visual-grounding","title":"Revisiting Visual Grounding","date":"2019-04-03","arxiv_id":"1904.02225","repositories_listed":0,"syntology":null},{"url":null,"slug":"align2ground-weakly-supervised-phrase","title":"Align2Ground: Weakly Supervised Phrase Grounding Guided by Image-Caption Alignment","date":"2019-03-27","arxiv_id":"1903.11649","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-only-look-listen-once-towards-fast-and","title":"You Only Look & Listen Once: Towards Fast and Accurate Visual Grounding","date":"2019-02-12","arxiv_id":"1902.04213","repositories_listed":0,"syntology":null},{"url":null,"slug":"taking-a-hint-leveraging-explanations-to-make","title":"Taking a HINT: Leveraging Explanations to Make Vision and Language Models More Grounded","date":"2019-02-11","arxiv_id":"1902.03751","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainability-by-parsing-neural-module-tree","title":"Learning to Assemble Neural Module Tree Networks for Visual Grounding","date":"2018-12-08","arxiv_id":"1812.03299","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-learning-of-hierarchical-vision","title":"Multi-task Learning of Hierarchical Vision-Language Representation","date":"2018-12-03","arxiv_id":"1812.00500","repositories_listed":0,"syntology":null},{"url":null,"slug":"being-data-driven-is-not-enough-revisiting","title":"Being data-driven is not enough: Revisiting interactive instruction giving as a challenge for NLG","date":"2018-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-language-priors-in-visual-question","title":"Overcoming Language Priors in Visual Question Answering with Adversarial Regularization","date":"2018-10-08","arxiv_id":"1810.03649","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-visual-question-answering-by-1","title":"Interpretable Visual Question Answering by Visual Grounding from Attention Supervision Mining","date":"2018-08-01","arxiv_id":"1808.00265","repositories_listed":0,"syntology":null},{"url":null,"slug":"illustrative-language-understanding-large","title":"Illustrative Language Understanding: Large-Scale Visual Grounding with Image Search","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"visually-grounded-cross-lingual-keyword","title":"Visually grounded cross-lingual keyword spotting in speech","date":"2018-06-13","arxiv_id":"1806.05030","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-visual-grounding-of-referring","title":"Interactive Visual Grounding of Referring Expressions for Human-Robot Interaction","date":"2018-06-11","arxiv_id":"1806.03831","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-it-weakly-supervised-reference-aware","title":"Finding \"It\": Weakly-Supervised Reference-Aware Visual Grounding in Instructional Videos","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-grounding-via-accumulated-attention","title":"Visual Grounding via Accumulated Attention","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-unsupervised-visual-grounding","title":"Learning Unsupervised Visual Grounding Through Semantic Self-Supervision","date":"2018-03-17","arxiv_id":"1803.06506","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-visually-grounded-sentence","title":"Improving Visually Grounded Sentence Representations with Self-Attention","date":"2017-12-02","arxiv_id":"1712.00609","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-reinforcement-learning-for-object","title":"Interactive Reinforcement Learning for Object Grounding via Self-Talking","date":"2017-12-02","arxiv_id":"1712.00576","repositories_listed":0,"syntology":null},{"url":"/paper/visual-reference-resolution-using-attention","slug":"visual-reference-resolution-using-attention","title":"Visual Reference Resolution using Attention Memory for Visual Dialog","date":"2017-09-23","arxiv_id":"1709.07992","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-visual-grounding-of-phrases","title":"Weakly-supervised Visual Grounding of Phrases with Linguistic Structures","date":"2017-05-03","arxiv_id":"1705.01371","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-grounded-conversations-multimodal","title":"Image-Grounded Conversations: Multimodal Context for Natural Question and Response Generation","date":"2017-01-28","arxiv_id":"1701.08251","repositories_listed":0,"syntology":null}],"record_sha256":"16fdbdf267ff5ce75c7a881bb4b69bfc98cc6b480abc5a8aaddae3f98450d3bc","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}