{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-grounding/papers/3","list_of":"/task/visual-grounding","task":"Visual Grounding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":6,"rows_per_page":100,"rows":[201,300],"of":571,"counts":{"archive_papers_tagged":571,"with_a_code_link":299,"where_syntology_ran_a_sample":111,"not_listed_spam_title":0,"listed":571,"listed_where_code_ran":111,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":95,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":95,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-grounding","prev":"/task/visual-grounding/papers/2","next":"/task/visual-grounding/papers/4","papers":[{"url":"/paper/language-adaptive-weight-generation-for-multi-1","slug":"language-adaptive-weight-generation-for-multi-1","title":"Language Adaptive Weight Generation for Multi-task Visual Grounding","date":"2023-06-06","arxiv_id":"2306.04652","repositories_listed":1,"syntology":null},{"url":"/paper/leverage-points-in-modality-shifts-comparing","slug":"leverage-points-in-modality-shifts-comparing","title":"Leverage Points in Modality Shifts: Comparing Language-only and Multimodal Word Representations","date":"2023-06-04","arxiv_id":"2306.02348","repositories_listed":1,"syntology":null},{"url":"/paper/an-examination-of-the-robustness-of-reference","slug":"an-examination-of-the-robustness-of-reference","title":"An Examination of the Robustness of Reference-Free Image Captioning Evaluation Metrics","date":"2023-05-24","arxiv_id":"2305.14998","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-examination-of-the-robustness-of-reference#ran","syntology_url":"https://syntology.ai/paper/2305.14998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14998"}},"official":{"repos":["saba96/img-cap-metrics-robustness"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/measuring-faithful-and-plausible-visual","slug":"measuring-faithful-and-plausible-visual","title":"Measuring Faithful and Plausible Visual Grounding in VQA","date":"2023-05-24","arxiv_id":"2305.15015","repositories_listed":1,"syntology":null},{"url":"/paper/cross3dvg-baseline-and-dataset-for-cross","slug":"cross3dvg-baseline-and-dataset-for-cross","title":"Cross3DVG: Cross-Dataset 3D Visual Grounding on Different RGB-D Scans","date":"2023-05-23","arxiv_id":"2305.13876","repositories_listed":1,"syntology":null},{"url":"/paper/wildrefer-3d-object-localization-in-large","slug":"wildrefer-3d-object-localization-in-large","title":"WildRefer: 3D Object Localization in Large-scale Dynamic Scenes with Multi-modal Visual Data and Natural Language","date":"2023-04-12","arxiv_id":"2304.05645","repositories_listed":1,"syntology":null},{"url":"/paper/scaneru-interactive-3d-visual-grounding-based","slug":"scaneru-interactive-3d-visual-grounding-based","title":"ScanERU: Interactive 3D Visual Grounding based on Embodied Reference Understanding","date":"2023-03-23","arxiv_id":"2303.13186","repositories_listed":1,"syntology":null},{"url":"/paper/joint-visual-grounding-and-tracking-with","slug":"joint-visual-grounding-and-tracking-with","title":"Joint Visual Grounding and Tracking with Natural Language Specification","date":"2023-03-21","arxiv_id":"2303.12027","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/joint-visual-grounding-and-tracking-with#ran","syntology_url":"https://syntology.ai/paper/2303.12027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.12027"}},"official":{"repos":["lizhou-cs/jointnlt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/champion-solution-for-the-wsdm2023-toloka-vqa","slug":"champion-solution-for-the-wsdm2023-toloka-vqa","title":"Champion Solution for the WSDM2023 Toloka VQA Challenge","date":"2023-01-22","arxiv_id":"2301.09045","repositories_listed":1,"syntology":null},{"url":"/paper/toward-building-general-foundation-models-for","slug":"toward-building-general-foundation-models-for","title":"Toward Building General Foundation Models for Language, Vision, and Vision-Language Understanding Tasks","date":"2023-01-12","arxiv_id":"2301.05065","repositories_listed":1,"syntology":null},{"url":"/paper/confidence-aware-pseudo-label-learning-for","slug":"confidence-aware-pseudo-label-learning-for","title":"Confidence-aware Pseudo-label Learning for Weakly Supervised Visual Grounding","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/context-aware-alignment-and-mutual-masking","slug":"context-aware-alignment-and-mutual-masking","title":"Context-Aware Alignment and Mutual Masking for 3D-Language Pre-Training","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/position-guided-text-prompt-for-vision","slug":"position-guided-text-prompt-for-vision","title":"Position-guided Text Prompt for Vision-Language Pre-training","date":"2022-12-19","arxiv_id":"2212.09737","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/position-guided-text-prompt-for-vision#ran","syntology_url":"https://syntology.ai/paper/2212.09737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09737"}},"official":{"repos":["sail-sg/ptp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dq-detr-dual-query-detection-transformer-for","slug":"dq-detr-dual-query-detection-transformer-for","title":"DQ-DETR: Dual Query Detection Transformer for Phrase Extraction and Grounding","date":"2022-11-28","arxiv_id":"2211.15516","repositories_listed":1,"syntology":null},{"url":"/paper/look-around-and-refer-2d-synthetic-semantics","slug":"look-around-and-refer-2d-synthetic-semantics","title":"Look Around and Refer: 2D Synthetic Semantics Knowledge Distillation for 3D Visual Grounding","date":"2022-11-25","arxiv_id":"2211.14241","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/look-around-and-refer-2d-synthetic-semantics#ran","syntology_url":"https://syntology.ai/paper/2211.14241","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.14241"}},"official":{"repos":["eslambakr/LAR-Look-Around-and-Refer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visually-grounded-vqa-by-lattice-based","slug":"visually-grounded-vqa-by-lattice-based","title":"Visually Grounded VQA by Lattice-based Retrieval","date":"2022-11-15","arxiv_id":"2211.08086","repositories_listed":1,"syntology":null},{"url":"/paper/yoro-lightweight-end-to-end-visual-grounding","slug":"yoro-lightweight-end-to-end-visual-grounding","title":"YORO -- Lightweight End to End Visual Grounding","date":"2022-11-15","arxiv_id":"2211.07912","repositories_listed":1,"syntology":null},{"url":"/paper/instruction-following-agents-with-jointly-pre","slug":"instruction-following-agents-with-jointly-pre","title":"Instruction-Following Agents with Multimodal Transformer","date":"2022-10-24","arxiv_id":"2210.13431","repositories_listed":1,"syntology":{"n":20,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/instruction-following-agents-with-jointly-pre#ran","syntology_url":"https://syntology.ai/paper/2210.13431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13431"}},"official":{"repos":["lhao499/instructrl"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/rsvg-exploring-data-and-models-for-visual","slug":"rsvg-exploring-data-and-models-for-visual","title":"RSVG: Exploring Data and Models for Visual Grounding on Remote Sensing Data","date":"2022-10-23","arxiv_id":"2210.12634","repositories_listed":1,"syntology":null},{"url":"/paper/ham-hierarchical-attention-model-with-high","slug":"ham-hierarchical-attention-model-with-high","title":"Learning Point-Language Hierarchical Alignment for 3D Visual Grounding","date":"2022-10-22","arxiv_id":"2210.12513","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ham-hierarchical-attention-model-with-high#ran","syntology_url":"https://syntology.ai/paper/2210.12513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.12513"}},"official":{"repos":["ppjmchen/ham"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-pre-training-basics-recent","slug":"vision-language-pre-training-basics-recent","title":"Vision-Language Pre-training: Basics, Recent Advances, and Future Trends","date":"2022-10-17","arxiv_id":"2210.09263","repositories_listed":1,"syntology":null},{"url":"/paper/a-hybrid-compositional-reasoning-approach-for","slug":"a-hybrid-compositional-reasoning-approach-for","title":"Enhancing Interpretability and Interactivity in Robot Manipulation: A Neurosymbolic Approach","date":"2022-10-03","arxiv_id":"2210.00858","repositories_listed":1,"syntology":null},{"url":"/paper/cost-effective-language-driven-image-editing","slug":"cost-effective-language-driven-image-editing","title":"Cost-Effective Language Driven Image Editing with LX-DRIM","date":"2022-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/gravl-bert-graphical-visual-linguistic","slug":"gravl-bert-graphical-visual-linguistic","title":"GRAVL-BERT: Graphical Visual-Linguistic Representations for Multimodal Coreference Resolution","date":"2022-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/introspective-learning-a-two-stage-approach-1","slug":"introspective-learning-a-two-stage-approach-1","title":"Introspective Learning : A Two-Stage Approach for Inference in Neural Networks","date":"2022-09-17","arxiv_id":"2209.08425","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/introspective-learning-a-two-stage-approach-1#ran","syntology_url":"https://syntology.ai/paper/2209.08425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.08425"}},"official":{"repos":["olivesgatech/introspective-learning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/efficient-vision-language-pretraining-with","slug":"efficient-vision-language-pretraining-with","title":"Efficient Vision-Language Pretraining with Visual Concepts and Hierarchical Alignment","date":"2022-08-29","arxiv_id":"2208.13628","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-vision-language-pretraining-with#ran","syntology_url":"https://syntology.ai/paper/2208.13628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.13628"}},"official":{"repos":["mshukor/vicha"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-grained-semantically-aligned-vision","slug":"fine-grained-semantically-aligned-vision","title":"Fine-Grained Semantically Aligned Vision-Language Pre-Training","date":"2022-08-04","arxiv_id":"2208.02515","repositories_listed":1,"syntology":null},{"url":"/paper/siri-a-simple-selective-retraining-mechanism","slug":"siri-a-simple-selective-retraining-mechanism","title":"SiRi: A Simple Selective Retraining Mechanism for Transformer-based Visual Grounding","date":"2022-07-27","arxiv_id":"2207.13325","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":2,"n_no_contract":3,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/siri-a-simple-selective-retraining-mechanism#ran","syntology_url":"https://syntology.ai/paper/2207.13325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.13325"}},"official":{"repos":["qumengxue/siri-vg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rovist-learning-robust-metrics-for-visual-2","slug":"rovist-learning-robust-metrics-for-visual-2","title":"RoViST: Learning Robust Metrics for Visual Storytelling","date":"2022-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/improving-visual-grounding-by-encouraging","slug":"improving-visual-grounding-by-encouraging","title":"Improving Visual Grounding by Encouraging Consistent Gradient-based Explanations","date":"2022-06-30","arxiv_id":"2206.15462","repositories_listed":1,"syntology":null},{"url":"/paper/language-with-vision-a-study-on-grounded-word","slug":"language-with-vision-a-study-on-grounded-word","title":"Language with Vision: a Study on Grounded Word and Sentence Embeddings","date":"2022-06-17","arxiv_id":"2206.08823","repositories_listed":1,"syntology":null},{"url":"/paper/mixgen-a-new-multi-modal-data-augmentation","slug":"mixgen-a-new-multi-modal-data-augmentation","title":"MixGen: A New Multi-Modal Data Augmentation","date":"2022-06-16","arxiv_id":"2206.08358","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/mixgen-a-new-multi-modal-data-augmentation#ran","syntology_url":"https://syntology.ai/paper/2206.08358","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08358"}},"official":{"repos":["amazon-research/mix-generation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/transvg-end-to-end-visual-grounding-with-1","slug":"transvg-end-to-end-visual-grounding-with-1","title":"TransVG++: End-to-End Visual Grounding with Language Conditioned Vision Transformer","date":"2022-06-14","arxiv_id":"2206.06619","repositories_listed":1,"syntology":null},{"url":"/paper/referring-image-matting","slug":"referring-image-matting","title":"Referring Image Matting","date":"2022-06-10","arxiv_id":"2206.05149","repositories_listed":1,"syntology":null},{"url":"/paper/rovist-learning-robust-metrics-for-visual-1","slug":"rovist-learning-robust-metrics-for-visual-1","title":"RoViST:Learning Robust Metrics for Visual Storytelling","date":"2022-05-08","arxiv_id":"2205.03774","repositories_listed":1,"syntology":null},{"url":"/paper/attention-as-grounding-exploring-textual-and-1","slug":"attention-as-grounding-exploring-textual-and-1","title":"Attention as Grounding: Exploring Textual and Cross-Modal Attention on Entities and Relations in Language-and-Vision Transformer","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/flexible-visual-grounding","slug":"flexible-visual-grounding","title":"Flexible Visual Grounding","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/to-find-waldo-you-need-contextual-cues-1","slug":"to-find-waldo-you-need-contextual-cues-1","title":"To Find Waldo You Need Contextual Cues: Debiasing Who’s Waldo","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/improving-visual-grounding-with-visual","slug":"improving-visual-grounding-with-visual","title":"Improving Visual Grounding with Visual-Linguistic Verification and Iterative Reasoning","date":"2022-04-30","arxiv_id":"2205.00272","repositories_listed":1,"syntology":null},{"url":"/paper/3d-sps-single-stage-3d-visual-grounding-via","slug":"3d-sps-single-stage-3d-visual-grounding-via","title":"3D-SPS: Single-Stage 3D Visual Grounding via Referred Point Progressive Selection","date":"2022-04-13","arxiv_id":"2204.06272","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/3d-sps-single-stage-3d-visual-grounding-via#ran","syntology_url":"https://syntology.ai/paper/2204.06272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.06272"}},"official":{"repos":["fjhzhixi/3d-sps"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-view-transformer-for-3d-visual","slug":"multi-view-transformer-for-3d-visual","title":"Multi-View Transformer for 3D Visual Grounding","date":"2022-04-05","arxiv_id":"2204.02174","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-view-transformer-for-3d-visual#ran","syntology_url":"https://syntology.ai/paper/2204.02174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02174"}},"official":{"repos":["sega-hsj/mvt-3dvg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/to-find-waldo-you-need-contextual-cues","slug":"to-find-waldo-you-need-contextual-cues","title":"To Find Waldo You Need Contextual Cues: Debiasing Who's Waldo","date":"2022-03-30","arxiv_id":"2203.16682","repositories_listed":1,"syntology":null},{"url":"/paper/tubedetr-spatio-temporal-video-grounding-with","slug":"tubedetr-spatio-temporal-video-grounding-with","title":"TubeDETR: Spatio-Temporal Video Grounding with Transformers","date":"2022-03-30","arxiv_id":"2203.16434","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/tubedetr-spatio-temporal-video-grounding-with#ran","syntology_url":"https://syntology.ai/paper/2203.16434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16434"}},"official":{"repos":["antoyang/TubeDETR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/shifting-more-attention-to-visual-backbone","slug":"shifting-more-attention-to-visual-backbone","title":"Shifting More Attention to Visual Backbone: Query-modulated Refinement Networks for End-to-End Visual Grounding","date":"2022-03-29","arxiv_id":"2203.15442","repositories_listed":1,"syntology":null},{"url":"/paper/local-global-context-aware-transformer-for","slug":"local-global-context-aware-transformer-for","title":"Local-Global Context Aware Transformer for Language-Guided Video Segmentation","date":"2022-03-18","arxiv_id":"2203.09773","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/local-global-context-aware-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2203.09773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09773"}},"official":{"repos":["leonnnop/locater"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pseudo-q-generating-pseudo-language-queries","slug":"pseudo-q-generating-pseudo-language-queries","title":"Pseudo-Q: Generating Pseudo Language Queries for Visual Grounding","date":"2022-03-16","arxiv_id":"2203.08481","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":7,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pseudo-q-generating-pseudo-language-queries#ran","syntology_url":"https://syntology.ai/paper/2203.08481","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08481"}},"official":{"repos":["leaplabthu/pseudo-q"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rex-reasoning-aware-and-grounded-explanation","slug":"rex-reasoning-aware-and-grounded-explanation","title":"REX: Reasoning-aware and Grounded Explanation","date":"2022-03-11","arxiv_id":"2203.06107","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rex-reasoning-aware-and-grounded-explanation#ran","syntology_url":"https://syntology.ai/paper/2203.06107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.06107"}},"official":{"repos":["szzexpoi/rex"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seeing-the-advantage-visually-grounding-word","slug":"seeing-the-advantage-visually-grounding-word","title":"Seeing the advantage: visually grounding word embeddings to better capture human semantic knowledge","date":"2022-02-21","arxiv_id":"2202.10292","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-representation-learning-for-8","slug":"self-supervised-representation-learning-for-8","title":"Self-Supervised Representation Learning for Speech Using Visual Grounding and Masked Language Modeling","date":"2022-02-07","arxiv_id":"2202.03543","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-dynamic-graph-transformer-for","slug":"multi-modal-dynamic-graph-transformer-for","title":"Multi-Modal Dynamic Graph Transformer for Visual Grounding","date":"2022-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/clip-lite-information-efficient-visual","slug":"clip-lite-information-efficient-visual","title":"CLIP-Lite: Information Efficient Visual Representation Learning with Language Supervision","date":"2021-12-14","arxiv_id":"2112.07133","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/clip-lite-information-efficient-visual#ran","syntology_url":"https://syntology.ai/paper/2112.07133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.07133"}},"official":{"repos":["4m4n5/CLIP-Lite"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/crossing-the-format-boundary-of-text-and","slug":"crossing-the-format-boundary-of-text-and","title":"UniTAB: Unifying Text and Box Outputs for Grounded Vision-Language Modeling","date":"2021-11-23","arxiv_id":"2111.12085","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":9,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/crossing-the-format-boundary-of-text-and#ran","syntology_url":"https://syntology.ai/paper/2111.12085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12085"}},"official":{"repos":["microsoft/UniTAB"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/grounded-situation-recognition-with","slug":"grounded-situation-recognition-with","title":"Grounded Situation Recognition with Transformers","date":"2021-11-19","arxiv_id":"2111.10135","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/grounded-situation-recognition-with#ran","syntology_url":"https://syntology.ai/paper/2111.10135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.10135"}},"official":{"repos":["jhcho99/gsrtr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-grained-vision-language-pre-training","slug":"multi-grained-vision-language-pre-training","title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","date":"2021-11-16","arxiv_id":"2111.08276","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-grained-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2111.08276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.08276"}},"official":{"repos":["zengyan-97/x-vlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-incremental-transformer-with","slug":"multimodal-incremental-transformer-with","title":"Multimodal Incremental Transformer with Visual Grounding for Visual Dialogue Generation","date":"2021-09-17","arxiv_id":"2109.08478","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-the-unknown-knowns-turning","slug":"discovering-the-unknown-knowns-turning","title":"Discovering the Unknown Knowns: Turning Implicit Knowledge in the Dataset into Explicit Training Examples for Visual Question Answering","date":"2021-09-13","arxiv_id":"2109.06122","repositories_listed":1,"syntology":null},{"url":"/paper/panoptic-narrative-grounding","slug":"panoptic-narrative-grounding","title":"Panoptic Narrative Grounding","date":"2021-09-10","arxiv_id":"2109.04988","repositories_listed":1,"syntology":null},{"url":"/paper/a-better-loss-for-visual-textual-grounding","slug":"a-better-loss-for-visual-textual-grounding","title":"A Better Loss for Visual-Textual Grounding","date":"2021-08-11","arxiv_id":"2108.05308","repositories_listed":1,"syntology":null},{"url":"/paper/vidlankd-improving-language-understanding-via","slug":"vidlankd-improving-language-understanding-via","title":"VidLanKD: Improving Language Understanding via Video-Distilled Knowledge Transfer","date":"2021-07-06","arxiv_id":"2107.02681","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-sentence-similarity-size-does-not","slug":"semantic-sentence-similarity-size-does-not","title":"Semantic sentence similarity: size does not always matter","date":"2021-06-16","arxiv_id":"2106.08648","repositories_listed":1,"syntology":null},{"url":"/paper/referring-transformer-a-one-step-approach-to","slug":"referring-transformer-a-one-step-approach-to","title":"Referring Transformer: A One-step Approach to Multi-task Visual Grounding","date":"2021-06-06","arxiv_id":"2106.03089","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/referring-transformer-a-one-step-approach-to#ran","syntology_url":"https://syntology.ai/paper/2106.03089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03089"}},"official":{"repos":["ubc-vision/RefTR"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-better-visual-dialog-agents-with","slug":"learning-better-visual-dialog-agents-with","title":"Learning Better Visual Dialog Agents with Pretrained Visual-Linguistic Representation","date":"2021-05-24","arxiv_id":"2105.11541","repositories_listed":1,"syntology":null},{"url":"/paper/sat-2d-semantics-assisted-training-for-3d","slug":"sat-2d-semantics-assisted-training-for-3d","title":"SAT: 2D Semantics Assisted Training for 3D Visual Grounding","date":"2021-05-24","arxiv_id":"2105.11450","repositories_listed":1,"syntology":null},{"url":"/paper/connecting-what-to-say-with-where-to-look-by","slug":"connecting-what-to-say-with-where-to-look-by","title":"Connecting What to Say With Where to Look by Modeling Human Attention Traces","date":"2021-05-12","arxiv_id":"2105.05964","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/connecting-what-to-say-with-where-to-look-by#ran","syntology_url":"https://syntology.ai/paper/2105.05964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.05964"}},"official":{"repos":["facebookresearch/connect-caption-and-trace"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/look-before-you-leap-learning-landmark","slug":"look-before-you-leap-learning-landmark","title":"Look Before You Leap: Learning Landmark Features for One-Stage Visual Grounding","date":"2021-04-09","arxiv_id":"2104.04386","repositories_listed":1,"syntology":null},{"url":"/paper/cyclic-co-learning-of-sounding-object-visual","slug":"cyclic-co-learning-of-sounding-object-visual","title":"Cyclic Co-Learning of Sounding Object Visual Grounding and Sound Separation","date":"2021-04-05","arxiv_id":"2104.02026","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cyclic-co-learning-of-sounding-object-visual#ran","syntology_url":"https://syntology.ai/paper/2104.02026","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.02026"}},"official":{"repos":["YapengTian/CCOL-CVPR21"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/relation-aware-instance-refinement-for-weakly","slug":"relation-aware-instance-refinement-for-weakly","title":"Relation-aware Instance Refinement for Weakly Supervised Visual Grounding","date":"2021-03-24","arxiv_id":"2103.12989","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/relation-aware-instance-refinement-for-weakly#ran","syntology_url":"https://syntology.ai/paper/2103.12989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.12989"}},"official":{"repos":["youngfly11/ReIR-WeaklyGrounding.pytorch"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ocid-ref-a-3d-robotic-dataset-with-embodied","slug":"ocid-ref-a-3d-robotic-dataset-with-embodied","title":"OCID-Ref: A 3D Robotic Dataset with Embodied Language for Clutter Scene Grounding","date":"2021-03-13","arxiv_id":"2103.07679","repositories_listed":1,"syntology":null},{"url":"/paper/instancerefer-cooperative-holistic","slug":"instancerefer-cooperative-holistic","title":"InstanceRefer: Cooperative Holistic Understanding for Visual Grounding on Point Clouds through Instance Multi-level Contextual Referring","date":"2021-03-01","arxiv_id":"2103.01128","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/instancerefer-cooperative-holistic#ran","syntology_url":"https://syntology.ai/paper/2103.01128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.01128"}},"official":{"repos":["CurryYuan/InstanceRefer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/answer-questions-with-right-image-regions-a","slug":"answer-questions-with-right-image-regions-a","title":"Answer Questions with Right Image Regions: A Visual Attention Regularization Approach","date":"2021-02-03","arxiv_id":"2102.01916","repositories_listed":1,"syntology":null},{"url":"/paper/panoptic-narrative-grounding-1","slug":"panoptic-narrative-grounding-1","title":"Panoptic Narrative Grounding","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/text-free-image-to-speech-synthesis-using","slug":"text-free-image-to-speech-synthesis-using","title":"Text-Free Image-to-Speech Synthesis Using Learned Segmental Units","date":"2020-12-31","arxiv_id":"2012.15454","repositories_listed":1,"syntology":null},{"url":"/paper/text-to-image-generation-grounded-by-fine","slug":"text-to-image-generation-grounded-by-fine","title":"Text-to-Image Generation Grounded by Fine-Grained User Attention","date":"2020-11-07","arxiv_id":"2011.03775","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-ground-medical-text-in-a-3d-human","slug":"learning-to-ground-medical-text-in-a-3d-human","title":"Learning to ground medical text in a 3D human atlas","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sort-ing-vqa-models-contrastive-gradient","slug":"sort-ing-vqa-models-contrastive-gradient","title":"SOrT-ing VQA Models : Contrastive Gradient Learning for Improved Consistency","date":"2020-10-20","arxiv_id":"2010.10038","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sort-ing-vqa-models-contrastive-gradient#ran","syntology_url":"https://syntology.ai/paper/2010.10038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10038"}},"official":{"repos":["sameerdharur/sorting-vqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-twins-talk","slug":"neural-twins-talk","title":"Neural Twins Talk","date":"2020-09-26","arxiv_id":"2009.12524","repositories_listed":1,"syntology":null},{"url":"/paper/cosine-meets-softmax-a-tough-to-beat-baseline","slug":"cosine-meets-softmax-a-tough-to-beat-baseline","title":"Cosine meets Softmax: A tough-to-beat baseline for visual grounding","date":"2020-09-13","arxiv_id":"2009.06066","repositories_listed":1,"syntology":null},{"url":"/paper/attngrounder-talking-to-cars-with-attention","slug":"attngrounder-talking-to-cars-with-attention","title":"AttnGrounder: Talking to Cars with Attention","date":"2020-09-11","arxiv_id":"2009.05684","repositories_listed":1,"syntology":null},{"url":"/paper/improving-one-stage-visual-grounding-by","slug":"improving-one-stage-visual-grounding-by","title":"Improving One-stage Visual Grounding by Recursive Sub-query Construction","date":"2020-08-03","arxiv_id":"2008.01059","repositories_listed":1,"syntology":{"n":22,"n_ran":20,"n_constructed":0,"n_ran_checked":19,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":18,"n_pointer_only":3,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 1 honoured, 0 violated, 18 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/improving-one-stage-visual-grounding-by#ran","syntology_url":"https://syntology.ai/paper/2008.01059","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.01059"}},"official":{"repos":["zyang-ur/ReSC"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":0,"n_ran_no_instrument_failure":19,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/spatially-aware-multimodal-transformers-for","slug":"spatially-aware-multimodal-transformers-for","title":"Spatially Aware Multimodal Transformers for TextVQA","date":"2020-07-23","arxiv_id":"2007.12146","repositories_listed":1,"syntology":null},{"url":"/paper/visual-relation-grounding-in-videos","slug":"visual-relation-grounding-in-videos","title":"Visual Relation Grounding in Videos","date":"2020-07-17","arxiv_id":"2007.08814","repositories_listed":1,"syntology":null},{"url":"/paper/improving-weakly-supervised-visual-grounding","slug":"improving-weakly-supervised-visual-grounding","title":"Improving Weakly Supervised Visual Grounding by Contrastive Knowledge Distillation","date":"2020-07-03","arxiv_id":"2007.01951","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-weakly-supervised-visual-grounding#ran","syntology_url":"https://syntology.ai/paper/2007.01951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.01951"}},"official":{"repos":["jhuang81/weak-sup-visual-grounding"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-grounding-of-learned-physical-models","slug":"visual-grounding-of-learned-physical-models","title":"Visual Grounding of Learned Physical Models","date":"2020-04-28","arxiv_id":"2004.13664","repositories_listed":1,"syntology":null},{"url":"/paper/deep-multimodal-neural-architecture-search","slug":"deep-multimodal-neural-architecture-search","title":"Deep Multimodal Neural Architecture Search","date":"2020-04-25","arxiv_id":"2004.12070","repositories_listed":1,"syntology":null},{"url":"/paper/a-negative-case-analysis-of-visual-grounding","slug":"a-negative-case-analysis-of-visual-grounding","title":"Visual Grounding Methods for VQA are Working for the Wrong Reasons!","date":"2020-04-12","arxiv_id":"2004.05704","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-negative-case-analysis-of-visual-grounding#ran","syntology_url":"https://syntology.ai/paper/2004.05704","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.05704"}},"official":{"repos":["erobic/negative_analysis_of_grounding"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-grounding-in-video-for-unsupervised","slug":"visual-grounding-in-video-for-unsupervised","title":"Visual Grounding in Video for Unsupervised Word Translation","date":"2020-03-11","arxiv_id":"2003.05078","repositories_listed":1,"syntology":null},{"url":"/paper/guessing-state-tracking-for-visual-dialogue","slug":"guessing-state-tracking-for-visual-dialogue","title":"Guessing State Tracking for Visual Dialogue","date":"2020-02-24","arxiv_id":"2002.10340","repositories_listed":1,"syntology":null},{"url":"/paper/learning-cross-modal-context-graph-for-visual","slug":"learning-cross-modal-context-graph-for-visual","title":"Learning Cross-modal Context Graph for Visual Grounding","date":"2020-02-13","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/connecting-vision-and-language-with-localized","slug":"connecting-vision-and-language-with-localized","title":"Connecting Vision and Language with Localized Narratives","date":"2019-12-06","arxiv_id":"1912.03098","repositories_listed":1,"syntology":null},{"url":"/paper/language-learning-using-speech-to-image","slug":"language-learning-using-speech-to-image","title":"Language learning using Speech to Image retrieval","date":"2019-09-09","arxiv_id":"1909.03795","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-query-by-example-speech-search-using","slug":"semantic-query-by-example-speech-search-using","title":"Semantic query-by-example speech search using visual grounding","date":"2019-04-15","arxiv_id":"1904.07078","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/semantic-query-by-example-speech-search-using#ran","syntology_url":"https://syntology.ai/paper/1904.07078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.07078"}},"official":null}},{"url":"/paper/modularized-textual-grounding-for","slug":"modularized-textual-grounding-for","title":"Modularized Textual Grounding for Counterfactual Resilience","date":"2019-04-07","arxiv_id":"1904.03589","repositories_listed":1,"syntology":null},{"url":"/paper/learning-semantic-sentence-representations","slug":"learning-semantic-sentence-representations","title":"Learning semantic sentence representations from visually grounded language without lexical knowledge","date":"2019-03-27","arxiv_id":"1903.11393","repositories_listed":1,"syntology":null},{"url":"/paper/visual-coreference-resolution-in-visual","slug":"visual-coreference-resolution-in-visual","title":"Visual Coreference Resolution in Visual Dialog using Neural Module Networks","date":"2018-09-06","arxiv_id":"1809.01816","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-diversified-and-discriminative","slug":"rethinking-diversified-and-discriminative","title":"Rethinking Diversified and Discriminative Proposal Generation for Visual Grounding","date":"2018-05-09","arxiv_id":"1805.03508","repositories_listed":1,"syntology":null},{"url":"/paper/finding-beans-in-burgers-deep-semantic-visual","slug":"finding-beans-in-burgers-deep-semantic-visual","title":"Finding beans in burgers: Deep semantic-visual embedding with localization","date":"2018-04-05","arxiv_id":"1804.01720","repositories_listed":1,"syntology":null},{"url":"/paper/self-view-grounding-given-a-narrated-360","slug":"self-view-grounding-given-a-narrated-360","title":"Self-view Grounding Given a Narrated 360° Video","date":"2017-11-23","arxiv_id":"1711.08664","repositories_listed":1,"syntology":null},{"url":"/paper/learning-two-branch-neural-networks-for-image","slug":"learning-two-branch-neural-networks-for-image","title":"Learning Two-Branch Neural Networks for Image-Text Matching Tasks","date":"2017-04-11","arxiv_id":"1704.03470","repositories_listed":1,"syntology":null},{"url":"/paper/visual-word2vec-vis-w2v-learning-visually","slug":"visual-word2vec-vis-w2v-learning-visually","title":"Visual Word2Vec (vis-w2v): Learning Visually Grounded Word Embeddings Using Abstract Scenes","date":"2015-11-22","arxiv_id":"1511.07067","repositories_listed":1,"syntology":null},{"url":null,"slug":"viewsrd-3d-visual-grounding-via-structured","title":"ViewSRD: 3D Visual Grounding via Structured Multi-View Decomposition","date":"2025-07-15","arxiv_id":"2507.11261","repositories_listed":0,"syntology":null}],"record_sha256":"e9fb2b689d8d7de7b84b9dfdba54f2ea3543b48c2ecd7db24e63b12e9d0177c4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}