{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-grounding/papers/2","list_of":"/task/visual-grounding","task":"Visual Grounding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":6,"rows_per_page":100,"rows":[101,200],"of":571,"counts":{"archive_papers_tagged":571,"with_a_code_link":299,"where_syntology_ran_a_sample":111,"not_listed_spam_title":0,"listed":571,"listed_where_code_ran":111,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":95,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":95,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-grounding","prev":"/task/visual-grounding","next":"/task/visual-grounding/papers/3","papers":[{"url":"/paper/griffon-g-bridging-vision-language-and-vision","slug":"griffon-g-bridging-vision-language-and-vision","title":"Griffon-G: Bridging Vision-Language and Vision-Centric Tasks via Large Multimodal Models","date":"2024-10-21","arxiv_id":"2410.16163","repositories_listed":1,"syntology":null},{"url":"/paper/vlm-grounder-a-vlm-agent-for-zero-shot-3d","slug":"vlm-grounder-a-vlm-agent-for-zero-shot-3d","title":"VLM-Grounder: A VLM Agent for Zero-Shot 3D Visual Grounding","date":"2024-10-17","arxiv_id":"2410.13860","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vlm-grounder-a-vlm-agent-for-zero-shot-3d#ran","syntology_url":"https://syntology.ai/paper/2410.13860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13860"}},"official":{"repos":["openrobotlab/vlm-grounder"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/context-infused-visual-grounding-for-art","slug":"context-infused-visual-grounding-for-art","title":"Context-Infused Visual Grounding for Art","date":"2024-10-16","arxiv_id":"2410.12369","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/context-infused-visual-grounding-for-art#ran","syntology_url":"https://syntology.ai/paper/2410.12369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12369"}},"official":{"repos":["selinakhan/CIGAr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mc-bench-a-benchmark-for-multi-context-visual","slug":"mc-bench-a-benchmark-for-multi-context-visual","title":"MC-Bench: A Benchmark for Multi-Context Visual Grounding in the Era of MLLMs","date":"2024-10-16","arxiv_id":"2410.12332","repositories_listed":1,"syntology":null},{"url":"/paper/vividmed-vision-language-model-with-versatile","slug":"vividmed-vision-language-model-with-versatile","title":"VividMed: Vision Language Model with Versatile Visual Grounding for Medicine","date":"2024-10-16","arxiv_id":"2410.12694","repositories_listed":1,"syntology":null},{"url":"/paper/videgothink-assessing-egocentric-video","slug":"videgothink-assessing-egocentric-video","title":"VidEgoThink: Assessing Egocentric Video Understanding Capabilities for Embodied AI","date":"2024-10-15","arxiv_id":"2410.11623","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":1,"n_honours":3,"n_violates":1,"n_no_contract":9,"n_pointer_only":5,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 1 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videgothink-assessing-egocentric-video#ran","syntology_url":"https://syntology.ai/paper/2410.11623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11623"}},"official":null}},{"url":"/paper/navigating-the-digital-world-as-humans-do","slug":"navigating-the-digital-world-as-humans-do","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","date":"2024-10-07","arxiv_id":"2410.05243","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/navigating-the-digital-world-as-humans-do#ran","syntology_url":"https://syntology.ai/paper/2410.05243","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05243"}},"official":{"repos":["OSU-NLP-Group/UGround"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/world-to-code-multi-modal-data-generation-via","slug":"world-to-code-multi-modal-data-generation-via","title":"World to Code: Multi-modal Data Generation via Self-Instructed Compositional Captioning and Filtering","date":"2024-09-30","arxiv_id":"2409.20424","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/world-to-code-multi-modal-data-generation-via#ran","syntology_url":"https://syntology.ai/paper/2409.20424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20424"}},"official":{"repos":["foundation-multimodal-models/world2code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/simvg-a-simple-framework-for-visual-grounding","slug":"simvg-a-simple-framework-for-visual-grounding","title":"SimVG: A Simple Framework for Visual Grounding with Decoupled Multi-modal Fusion","date":"2024-09-26","arxiv_id":"2409.17531","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/simvg-a-simple-framework-for-visual-grounding#ran","syntology_url":"https://syntology.ai/paper/2409.17531","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.17531"}},"official":{"repos":["dmmm1997/simvg"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hifi-cs-towards-open-vocabulary-visual","slug":"hifi-cs-towards-open-vocabulary-visual","title":"HiFi-CS: Towards Open Vocabulary Visual Grounding For Robotic Grasping Using Vision-Language Models","date":"2024-09-16","arxiv_id":"2409.10419","repositories_listed":1,"syntology":null},{"url":"/paper/shaking-up-vlms-comparing-transformers-and","slug":"shaking-up-vlms-comparing-transformers-and","title":"Shaking Up VLMs: Comparing Transformers and Structured State Space Models for Vision & Language Modeling","date":"2024-09-09","arxiv_id":"2409.05395","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shaking-up-vlms-comparing-transformers-and#ran","syntology_url":"https://syntology.ai/paper/2409.05395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.05395"}},"official":{"repos":["gpantaz/vl_mamba"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-grounding-with-multi-modal-conditional","slug":"visual-grounding-with-multi-modal-conditional","title":"Visual Grounding with Multi-modal Conditional Adaptation","date":"2024-09-08","arxiv_id":"2409.04999","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visual-grounding-with-multi-modal-conditional#ran","syntology_url":"https://syntology.ai/paper/2409.04999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.04999"}},"official":{"repos":["mr-bigworth/mmca"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lexicon3d-probing-visual-foundation-models","slug":"lexicon3d-probing-visual-foundation-models","title":"Lexicon3D: Probing Visual Foundation Models for Complex 3D Scene Understanding","date":"2024-09-05","arxiv_id":"2409.03757","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lexicon3d-probing-visual-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2409.03757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.03757"}},"official":null}},{"url":"/paper/resvg-enhancing-relation-and-semantic","slug":"resvg-enhancing-relation-and-semantic","title":"ResVG: Enhancing Relation and Semantic Understanding in Multiple Instances for Visual Grounding","date":"2024-08-29","arxiv_id":"2408.16314","repositories_listed":1,"syntology":null},{"url":"/paper/iaa-inner-adaptor-architecture-empowers","slug":"iaa-inner-adaptor-architecture-empowers","title":"IAA: Inner-Adaptor Architecture Empowers Frozen Large Language Model with Multimodal Capabilities","date":"2024-08-23","arxiv_id":"2408.12902","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/iaa-inner-adaptor-architecture-empowers#ran","syntology_url":"https://syntology.ai/paper/2408.12902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12902"}},"official":{"repos":["360cvgroup/inner-adaptor-architecture"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/in-defense-of-lazy-visual-grounding-for-open","slug":"in-defense-of-lazy-visual-grounding-for-open","title":"In Defense of Lazy Visual Grounding for Open-Vocabulary Semantic Segmentation","date":"2024-08-09","arxiv_id":"2408.04961","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01942","slug":"2408-01942","title":"Visual Grounding for Object-Level Generalization in Reinforcement Learning","date":"2024-08-04","arxiv_id":"2408.01942","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/2408-01942#ran","syntology_url":"https://syntology.ai/paper/2408.01942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01942"}},"official":{"repos":["pku-rl/copl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-01120","slug":"2408-01120","title":"An Efficient and Effective Transformer Decoder-Based Framework for Multi-Task Visual Grounding","date":"2024-08-02","arxiv_id":"2408.01120","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/2408-01120#ran","syntology_url":"https://syntology.ai/paper/2408.01120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01120"}},"official":{"repos":["chenwei746/eevg"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/refmask3d-language-guided-transformer-for-3d","slug":"refmask3d-language-guided-transformer-for-3d","title":"RefMask3D: Language-Guided Transformer for 3D Referring Segmentation","date":"2024-07-25","arxiv_id":"2407.18244","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/refmask3d-language-guided-transformer-for-3d#ran","syntology_url":"https://syntology.ai/paper/2407.18244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18244"}},"official":{"repos":["heshuting555/refmask3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/3d-vision-and-language-pretraining-with-large","slug":"3d-vision-and-language-pretraining-with-large","title":"3D Vision and Language Pretraining with Large-Scale Synthetic Data","date":"2024-07-08","arxiv_id":"2407.06084","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/3d-vision-and-language-pretraining-with-large#ran","syntology_url":"https://syntology.ai/paper/2407.06084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06084"}},"official":{"repos":["idejie/3DSyn"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-phrase-level-grounding-with-text-to","slug":"exploring-phrase-level-grounding-with-text-to","title":"Exploring Phrase-Level Grounding with Text-to-Image Diffusion Model","date":"2024-07-07","arxiv_id":"2407.05352","repositories_listed":1,"syntology":null},{"url":"/paper/multi-branch-collaborative-learning-network","slug":"multi-branch-collaborative-learning-network","title":"Multi-branch Collaborative Learning Network for 3D Visual Grounding","date":"2024-07-07","arxiv_id":"2407.05363","repositories_listed":1,"syntology":null},{"url":"/paper/not-yet-the-whole-story-evaluating-visual","slug":"not-yet-the-whole-story-evaluating-visual","title":"Not (yet) the whole story: Evaluating Visual Storytelling Requires More than Measuring Coherence, Grounding, and Repetition","date":"2024-07-05","arxiv_id":"2407.04559","repositories_listed":1,"syntology":null},{"url":"/paper/smart-vision-language-reasoners","slug":"smart-vision-language-reasoners","title":"Smart Vision-Language Reasoners","date":"2024-07-05","arxiv_id":"2407.04212","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/smart-vision-language-reasoners#ran","syntology_url":"https://syntology.ai/paper/2407.04212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04212"}},"official":{"repos":["smarter-vlm/smarter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/segvg-transferring-object-bounding-box-to","slug":"segvg-transferring-object-bounding-box-to","title":"SegVG: Transferring Object Bounding Box to Segmentation for Visual Grounding","date":"2024-07-03","arxiv_id":"2407.03200","repositories_listed":1,"syntology":{"n":31,"n_ran":20,"n_constructed":15,"n_ran_checked":15,"n_instrument":5,"n_unverified":11,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":31,"phrase":"20 ran (of which 15 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 5 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/segvg-transferring-object-bounding-box-to#ran","syntology_url":"https://syntology.ai/paper/2407.03200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03200"}},"official":{"repos":["weitaikang/segvg"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":15,"n_ran_no_instrument_failure":15,"n_unverified":11,"ran_from_kinds":["official"]}}},{"url":"/paper/cvlue-a-new-benchmark-dataset-for-chinese","slug":"cvlue-a-new-benchmark-dataset-for-chinese","title":"CVLUE: A New Benchmark Dataset for Chinese Vision-Language Understanding Evaluation","date":"2024-07-01","arxiv_id":"2407.01081","repositories_listed":1,"syntology":null},{"url":"/paper/cambrian-1-a-fully-open-vision-centric","slug":"cambrian-1-a-fully-open-vision-centric","title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","date":"2024-06-24","arxiv_id":"2406.16860","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":3,"n_ran_checked":5,"n_instrument":5,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cambrian-1-a-fully-open-vision-centric#ran","syntology_url":"https://syntology.ai/paper/2406.16860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16860"}},"official":{"repos":["cambrian-mllm/cambrian"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/agla-mitigating-object-hallucinations-in","slug":"agla-mitigating-object-hallucinations-in","title":"AGLA: Mitigating Object Hallucinations in Large Vision-Language Models with Assembly of Global and Local Attention","date":"2024-06-18","arxiv_id":"2406.12718","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/agla-mitigating-object-hallucinations-in#ran","syntology_url":"https://syntology.ai/paper/2406.12718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12718"}},"official":{"repos":["lackel/agla"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mmscan-a-multi-modal-3d-scene-dataset-with","slug":"mmscan-a-multi-modal-3d-scene-dataset-with","title":"MMScan: A Multi-Modal 3D Scene Dataset with Hierarchical Grounded Language Annotations","date":"2024-06-13","arxiv_id":"2406.09401","repositories_listed":1,"syntology":null},{"url":"/paper/towards-vision-language-geo-foundation-model","slug":"towards-vision-language-geo-foundation-model","title":"Towards Vision-Language Geo-Foundation Model: A Survey","date":"2024-06-13","arxiv_id":"2406.09385","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-text-guided-3d-visual-grounding","slug":"a-survey-on-text-guided-3d-visual-grounding","title":"A Survey on Text-guided 3D Visual Grounding: Elements, Recent Advances, and Future Directions","date":"2024-06-09","arxiv_id":"2406.05785","repositories_listed":1,"syntology":null},{"url":"/paper/separating-the-chirp-from-the-chat-self","slug":"separating-the-chirp-from-the-chat-self","title":"Separating the \"Chirp\" from the \"Chat\": Self-supervised Visual Grounding of Sound and Language","date":"2024-06-09","arxiv_id":"2406.05629","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/separating-the-chirp-from-the-chat-self#ran","syntology_url":"https://syntology.ai/paper/2406.05629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05629"}},"official":{"repos":["mhamilton723/DenseAV"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/instruction-guided-visual-masking","slug":"instruction-guided-visual-masking","title":"Instruction-Guided Visual Masking","date":"2024-05-30","arxiv_id":"2405.19783","repositories_listed":1,"syntology":{"n":28,"n_ran":20,"n_constructed":7,"n_ran_checked":11,"n_instrument":9,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"20 ran (of which 7 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 9 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/instruction-guided-visual-masking#ran","syntology_url":"https://syntology.ai/paper/2405.19783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19783"}},"official":{"repos":["2toinf/ivm"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":7,"n_ran_no_instrument_failure":11,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/talk2radar-bridging-natural-language-with-4d","slug":"talk2radar-bridging-natural-language-with-4d","title":"Talk2Radar: Bridging Natural Language with 4D mmWave Radar for 3D Referring Expression Comprehension","date":"2024-05-21","arxiv_id":"2405.12821","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-robustness-for-visual-grounding","slug":"adversarial-robustness-for-visual-grounding","title":"Adversarial Robustness for Visual Grounding of Multimodal Large Language Models","date":"2024-05-16","arxiv_id":"2405.09981","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarial-robustness-for-visual-grounding#ran","syntology_url":"https://syntology.ai/paper/2405.09981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.09981"}},"official":{"repos":["KuofengGao/MLLM-Grounding-Robustness"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dara-domain-and-relation-aware-adapters-make","slug":"dara-domain-and-relation-aware-adapters-make","title":"DARA: Domain- and Relation-aware Adapters Make Parameter-efficient Tuning for Visual Grounding","date":"2024-05-10","arxiv_id":"2405.06217","repositories_listed":1,"syntology":null},{"url":"/paper/list-items-one-by-one-a-new-data-source-and","slug":"list-items-one-by-one-a-new-data-source-and","title":"List Items One by One: A New Data Source and Learning Paradigm for Multimodal LLMs","date":"2024-04-25","arxiv_id":"2404.16375","repositories_listed":1,"syntology":null},{"url":"/paper/hivg-hierarchical-multimodal-fine-grained","slug":"hivg-hierarchical-multimodal-fine-grained","title":"HiVG: Hierarchical Multimodal Fine-grained Modulation for Visual Grounding","date":"2024-04-20","arxiv_id":"2404.13400","repositories_listed":1,"syntology":null},{"url":"/paper/groma-localized-visual-tokenization-for","slug":"groma-localized-visual-tokenization-for","title":"Groma: Localized Visual Tokenization for Grounding Multimodal Large Language Models","date":"2024-04-19","arxiv_id":"2404.13013","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/groma-localized-visual-tokenization-for#ran","syntology_url":"https://syntology.ai/paper/2404.13013","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13013"}},"official":{"repos":["FoundationVision/Groma"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-3d-dense-caption-and-visual","slug":"rethinking-3d-dense-caption-and-visual","title":"Rethinking 3D Dense Caption and Visual Grounding in A Unified Framework through Prompt-based Localization","date":"2024-04-17","arxiv_id":"2404.11064","repositories_listed":1,"syntology":null},{"url":"/paper/agentstudio-a-toolkit-for-building-general","slug":"agentstudio-a-toolkit-for-building-general","title":"AgentStudio: A Toolkit for Building General Virtual Agents","date":"2024-03-26","arxiv_id":"2403.17918","repositories_listed":1,"syntology":null},{"url":"/paper/medpromptx-grounded-multimodal-prompting-for","slug":"medpromptx-grounded-multimodal-prompting-for","title":"MedPromptX: Grounded Multimodal Prompting for Chest X-ray Diagnosis","date":"2024-03-22","arxiv_id":"2403.15585","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-transferability-in-vision-language","slug":"boosting-transferability-in-vision-language","title":"Boosting Transferability in Vision-Language Attacks via Diversification along the Intersection Region of Adversarial Trajectory","date":"2024-03-19","arxiv_id":"2403.12445","repositories_listed":1,"syntology":null},{"url":"/paper/hydra-a-hyper-agent-for-dynamic-compositional","slug":"hydra-a-hyper-agent-for-dynamic-compositional","title":"HYDRA: A Hyper Agent for Dynamic Compositional Visual Reasoning","date":"2024-03-19","arxiv_id":"2403.12884","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hydra-a-hyper-agent-for-dynamic-compositional#ran","syntology_url":"https://syntology.ai/paper/2403.12884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12884"}},"official":{"repos":["ControlNet/HYDRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/secg-semantic-enhanced-3d-visual-grounding","slug":"secg-semantic-enhanced-3d-visual-grounding","title":"SeCG: Semantic-Enhanced 3D Visual Grounding via Cross-modal Graph Attention","date":"2024-03-13","arxiv_id":"2403.08182","repositories_listed":1,"syntology":null},{"url":"/paper/mikasa-multi-key-anchor-scene-aware","slug":"mikasa-multi-key-anchor-scene-aware","title":"MiKASA: Multi-Key-Anchor & Scene-Aware Transformer for 3D Visual Grounding","date":"2024-03-05","arxiv_id":"2403.03077","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":10,"n_ran_checked":11,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":16,"phrase":"11 ran (of which 10 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mikasa-multi-key-anchor-scene-aware#ran","syntology_url":"https://syntology.ai/paper/2403.03077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03077"}},"official":{"repos":["dfki-av/mikasa-3dvg"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":10,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/the-r-evolution-of-multimodal-large-language","slug":"the-r-evolution-of-multimodal-large-language","title":"The Revolution of Multimodal Large Language Models: A Survey","date":"2024-02-19","arxiv_id":"2402.12451","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-literal-descriptions-understanding-and","slug":"beyond-literal-descriptions-understanding-and","title":"Beyond Literal Descriptions: Understanding and Locating Open-World Objects Aligned with Human Intentions","date":"2024-02-17","arxiv_id":"2402.11265","repositories_listed":1,"syntology":null},{"url":"/paper/vigor-improving-visual-grounding-of-large","slug":"vigor-improving-visual-grounding-of-large","title":"ViGoR: Improving Visual Grounding of Large Vision Language Models with Fine-Grained Reward Modeling","date":"2024-02-09","arxiv_id":"2402.06118","repositories_listed":1,"syntology":null},{"url":"/paper/chatterbox-multi-round-multimodal-referring","slug":"chatterbox-multi-round-multimodal-referring","title":"ChatterBox: Multi-round Multimodal Referring and Grounding","date":"2024-01-24","arxiv_id":"2401.13307","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-visual-and-vision-language-tracking","slug":"unifying-visual-and-vision-language-tracking","title":"Unifying Visual and Vision-Language Tracking via Contrastive Learning","date":"2024-01-20","arxiv_id":"2401.11228","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-visual-and-vision-language-tracking#ran","syntology_url":"https://syntology.ai/paper/2401.11228","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11228"}},"official":{"repos":["openspaceai/uvltrack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/skyeyegpt-unifying-remote-sensing-vision","slug":"skyeyegpt-unifying-remote-sensing-vision","title":"SkyEyeGPT: Unifying Remote Sensing Vision-Language Tasks via Instruction Tuning with Large Language Model","date":"2024-01-18","arxiv_id":"2401.09712","repositories_listed":1,"syntology":null},{"url":"/paper/veagle-advancements-in-multimodal","slug":"veagle-advancements-in-multimodal","title":"Veagle: Advancements in Multimodal Representation Learning","date":"2024-01-18","arxiv_id":"2403.08773","repositories_listed":1,"syntology":null},{"url":"/paper/uncovering-the-full-potential-of-visual","slug":"uncovering-the-full-potential-of-visual","title":"Uncovering the Full Potential of Visual Grounding Methods in VQA","date":"2024-01-15","arxiv_id":"2401.07803","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uncovering-the-full-potential-of-visual#ran","syntology_url":"https://syntology.ai/paper/2401.07803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07803"}},"official":{"repos":["dreichcsl/truevg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/eyes-wide-shut-exploring-the-visual","slug":"eyes-wide-shut-exploring-the-visual","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","date":"2024-01-11","arxiv_id":"2401.06209","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eyes-wide-shut-exploring-the-visual#ran","syntology_url":"https://syntology.ai/paper/2401.06209","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06209"}},"official":{"repos":["tsb0601/MMVP"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/investigating-compositional-challenges-in","slug":"investigating-compositional-challenges-in","title":"Investigating Compositional Challenges in Vision-Language Models for Visual Grounding","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-attribute-interactions-matter-for-3d","slug":"multi-attribute-interactions-matter-for-3d","title":"Multi-Attribute Interactions Matter for 3D Visual Grounding","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-parts-beyond-objects-towards-finer-1","slug":"unveiling-parts-beyond-objects-towards-finer-1","title":"Unveiling Parts Beyond Objects: Towards Finer-Granularity Referring Expression Segmentation","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/v-guided-visual-search-as-a-core-mechanism-in","slug":"v-guided-visual-search-as-a-core-mechanism-in","title":"V?: Guided Visual Search as a Core Mechanism in Multimodal LLMs","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/groundvlp-harnessing-zero-shot-visual","slug":"groundvlp-harnessing-zero-shot-visual","title":"GroundVLP: Harnessing Zero-shot Visual Grounding from Vision-Language Pre-training and Open-Vocabulary Object Detection","date":"2023-12-22","arxiv_id":"2312.15043","repositories_listed":1,"syntology":null},{"url":"/paper/context-disentangling-and-prototype","slug":"context-disentangling-and-prototype","title":"Context Disentangling and Prototype Inheriting for Robust Visual Grounding","date":"2023-12-19","arxiv_id":"2312.11967","repositories_listed":1,"syntology":null},{"url":"/paper/mask-grounding-for-referring-image","slug":"mask-grounding-for-referring-image","title":"Mask Grounding for Referring Image Segmentation","date":"2023-12-19","arxiv_id":"2312.12198","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":5,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mask-grounding-for-referring-image#ran","syntology_url":"https://syntology.ai/paper/2312.12198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12198"}},"official":{"repos":["yxchng/mask-grounding"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mono3dvg-3d-visual-grounding-in-monocular","slug":"mono3dvg-3d-visual-grounding-in-monocular","title":"Mono3DVG: 3D Visual Grounding in Monocular Images","date":"2023-12-13","arxiv_id":"2312.08022","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mono3dvg-3d-visual-grounding-in-monocular#ran","syntology_url":"https://syntology.ai/paper/2312.08022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08022"}},"official":{"repos":["zhanyang-nwpu/mono3dvg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unveiling-parts-beyond-objects-towards-finer","slug":"unveiling-parts-beyond-objects-towards-finer","title":"Unveiling Parts Beyond Objects:Towards Finer-Granularity Referring Expression Segmentation","date":"2023-12-13","arxiv_id":"2312.08007","repositories_listed":1,"syntology":null},{"url":"/paper/gpt-4-enhanced-multimodal-grounding-for","slug":"gpt-4-enhanced-multimodal-grounding-for","title":"GPT-4 Enhanced Multimodal Grounding for Autonomous Driving: Leveraging Cross-Modal Attention with Large Language Models","date":"2023-12-06","arxiv_id":"2312.03543","repositories_listed":1,"syntology":null},{"url":"/paper/mismatch-quest-visual-and-textual-feedback","slug":"mismatch-quest-visual-and-textual-feedback","title":"Mismatch Quest: Visual and Textual Feedback for Image-Text Misalignment","date":"2023-12-05","arxiv_id":"2312.03766","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mismatch-quest-visual-and-textual-feedback#ran","syntology_url":"https://syntology.ai/paper/2312.03766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03766"}},"official":{"repos":["mismatchquest/mismatchquest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/behind-the-magic-merlim-multi-modal","slug":"behind-the-magic-merlim-multi-modal","title":"Behind the Magic, MERLIM: Multi-modal Evaluation Benchmark for Large Image-Language Models","date":"2023-12-03","arxiv_id":"2312.02219","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/behind-the-magic-merlim-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2312.02219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02219"}},"official":{"repos":["ojedaf/merlim"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/g2d-from-global-to-dense-radiography","slug":"g2d-from-global-to-dense-radiography","title":"G2D: From Global to Dense Radiography Representation Learning via Vision-Language Pre-training","date":"2023-12-03","arxiv_id":"2312.01522","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/g2d-from-global-to-dense-radiography#ran","syntology_url":"https://syntology.ai/paper/2312.01522","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01522"}},"official":{"repos":["cheliu-computation/g2d-neurips24"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-referring-expression-comprehension","slug":"zero-shot-referring-expression-comprehension","title":"Zero-shot Referring Expression Comprehension via Structural Similarity Between Images and Captions","date":"2023-11-28","arxiv_id":"2311.17048","repositories_listed":1,"syntology":{"n":20,"n_ran":14,"n_constructed":0,"n_ran_checked":10,"n_instrument":4,"n_unverified":6,"n_honours":0,"n_violates":3,"n_no_contract":7,"n_pointer_only":10,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 3 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/zero-shot-referring-expression-comprehension#ran","syntology_url":"https://syntology.ai/paper/2311.17048","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17048"}},"official":{"repos":["show-han/zeroshot_rec"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-programming-for-zero-shot-open","slug":"visual-programming-for-zero-shot-open","title":"Visual Programming for Zero-shot Open-Vocabulary 3D Visual Grounding","date":"2023-11-26","arxiv_id":"2311.15383","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-programming-for-zero-shot-open#ran","syntology_url":"https://syntology.ai/paper/2311.15383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15383"}},"official":{"repos":["CurryYuan/ZSVG3D"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/florence-2-advancing-a-unified-representation","slug":"florence-2-advancing-a-unified-representation","title":"Florence-2: Advancing a Unified Representation for a Variety of Vision Tasks","date":"2023-11-10","arxiv_id":"2311.06242","repositories_listed":1,"syntology":null},{"url":"/paper/language-guided-robot-grasping-clip-based","slug":"language-guided-robot-grasping-clip-based","title":"Language-guided Robot Grasping: CLIP-based Referring Grasp Synthesis in Clutter","date":"2023-11-09","arxiv_id":"2311.05779","repositories_listed":1,"syntology":null},{"url":"/paper/next-chat-an-lmm-for-chat-detection-and","slug":"next-chat-an-lmm-for-chat-detection-and","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","date":"2023-11-08","arxiv_id":"2311.04498","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/next-chat-an-lmm-for-chat-detection-and#ran","syntology_url":"https://syntology.ai/paper/2311.04498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04498"}},"official":{"repos":["next-chatv/next-chat"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-grounding-potential-of-vqa-oriented","slug":"exploring-grounding-potential-of-vqa-oriented","title":"GPT-4V-AD: Exploring Grounding Potential of VQA-oriented GPT-4V for Zero-shot Anomaly Detection","date":"2023-11-05","arxiv_id":"2311.02612","repositories_listed":1,"syntology":null},{"url":"/paper/cityrefer-geography-aware-3d-visual-grounding","slug":"cityrefer-geography-aware-3d-visual-grounding","title":"CityRefer: Geography-aware 3D Visual Grounding Dataset on City-scale Point Cloud Data","date":"2023-10-28","arxiv_id":"2310.18773","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cityrefer-geography-aware-3d-visual-grounding#ran","syntology_url":"https://syntology.ai/paper/2310.18773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18773"}},"official":{"repos":["atr-dbi/cityrefer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/groovist-a-metric-for-grounding-objects-in","slug":"groovist-a-metric-for-grounding-objects-in","title":"GROOViST: A Metric for Grounding Objects in Visual Storytelling","date":"2023-10-26","arxiv_id":"2310.17770","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/groovist-a-metric-for-grounding-objects-in#ran","syntology_url":"https://syntology.ai/paper/2310.17770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17770"}},"official":{"repos":["akskuchi/groovist"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ov-vg-a-benchmark-for-open-vocabulary-visual","slug":"ov-vg-a-benchmark-for-open-vocabulary-visual","title":"OV-VG: A Benchmark for Open-Vocabulary Visual Grounding","date":"2023-10-22","arxiv_id":"2310.14374","repositories_listed":1,"syntology":null},{"url":"/paper/invig-benchmarking-interactive-visual","slug":"invig-benchmarking-interactive-visual","title":"InViG: Benchmarking Interactive Visual Grounding with 500K Human-Robot Interactions","date":"2023-10-18","arxiv_id":"2310.12147","repositories_listed":1,"syntology":null},{"url":"/paper/nice-improving-panoptic-narrative-detection","slug":"nice-improving-panoptic-narrative-detection","title":"NICE: Improving Panoptic Narrative Detection and Segmentation with Cascading Collaborative Learning","date":"2023-10-17","arxiv_id":"2310.10975","repositories_listed":1,"syntology":null},{"url":"/paper/from-clip-to-dino-visual-encoders-shout-in","slug":"from-clip-to-dino-visual-encoders-shout-in","title":"From CLIP to DINO: Visual Encoders Shout in Multi-modal Large Language Models","date":"2023-10-13","arxiv_id":"2310.08825","repositories_listed":1,"syntology":null},{"url":"/paper/cot3dref-chain-of-thoughts-data-efficient-3d","slug":"cot3dref-chain-of-thoughts-data-efficient-3d","title":"CoT3DRef: Chain-of-Thoughts Data-Efficient 3D Visual Grounding","date":"2023-10-10","arxiv_id":"2310.06214","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cot3dref-chain-of-thoughts-data-efficient-3d#ran","syntology_url":"https://syntology.ai/paper/2310.06214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06214"}},"official":{"repos":["eslambakr/CoT3D_VG"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rephrase-augment-reason-visual-grounding-of","slug":"rephrase-augment-reason-visual-grounding-of","title":"Rephrase, Augment, Reason: Visual Grounding of Questions for Vision-Language Models","date":"2023-10-09","arxiv_id":"2310.05861","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rephrase-augment-reason-visual-grounding-of#ran","syntology_url":"https://syntology.ai/paper/2310.05861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05861"}},"official":{"repos":["archiki/repare"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-grounder-open-vocabulary-3d-visual","slug":"llm-grounder-open-vocabulary-3d-visual","title":"LLM-Grounder: Open-Vocabulary 3D Visual Grounding with Large Language Model as an Agent","date":"2023-09-21","arxiv_id":"2309.12311","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llm-grounder-open-vocabulary-3d-visual#ran","syntology_url":"https://syntology.ai/paper/2309.12311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12311"}},"official":{"repos":["sled-group/chat-with-nerf"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prograsp-pragmatic-human-robot-communication","slug":"prograsp-pragmatic-human-robot-communication","title":"PROGrasp: Pragmatic Human-Robot Communication for Object Grasping","date":"2023-09-14","arxiv_id":"2309.07759","repositories_listed":1,"syntology":null},{"url":"/paper/multi3drefer-grounding-text-description-to","slug":"multi3drefer-grounding-text-description-to","title":"Multi3DRefer: Grounding Text Description to Multiple 3D Objects","date":"2023-09-11","arxiv_id":"2309.05251","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multi3drefer-grounding-text-description-to#ran","syntology_url":"https://syntology.ai/paper/2309.05251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05251"}},"official":{"repos":["3dlg-hcvc/M3DRef-CLIP"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/collecting-visually-grounded-dialogue-with-a-1","slug":"collecting-visually-grounded-dialogue-with-a-1","title":"Collecting Visually-Grounded Dialogue with A Game Of Sorts","date":"2023-09-10","arxiv_id":"2309.05162","repositories_listed":1,"syntology":null},{"url":"/paper/determinet-a-large-scale-diagnostic-dataset","slug":"determinet-a-large-scale-diagnostic-dataset","title":"DetermiNet: A Large-Scale Diagnostic Dataset for Complex Visually-Grounded Referencing using Determiners","date":"2023-09-07","arxiv_id":"2309.03483","repositories_listed":1,"syntology":null},{"url":"/paper/vgdiffzero-text-to-image-diffusion-models-can","slug":"vgdiffzero-text-to-image-diffusion-models-can","title":"VGDiffZero: Text-to-image Diffusion Models Can Be Zero-shot Visual Grounders","date":"2023-09-03","arxiv_id":"2309.01141","repositories_listed":1,"syntology":null},{"url":"/paper/unipt-universal-parallel-tuning-for-transfer","slug":"unipt-universal-parallel-tuning-for-transfer","title":"UniPT: Universal Parallel Tuning for Transfer Learning with Efficient Parameter and Memory","date":"2023-08-28","arxiv_id":"2308.14316","repositories_listed":1,"syntology":null},{"url":"/paper/hubo-vlm-unified-vision-language-model","slug":"hubo-vlm-unified-vision-language-model","title":"HuBo-VLM: Unified Vision-Language Model designed for HUman roBOt interaction tasks","date":"2023-08-24","arxiv_id":"2308.12537","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-framework-for-3d-point-cloud-visual","slug":"a-unified-framework-for-3d-point-cloud-visual","title":"A Unified Framework for 3D Point Cloud Visual Grounding","date":"2023-08-23","arxiv_id":"2308.11887","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-framework-for-3d-point-cloud-visual#ran","syntology_url":"https://syntology.ai/paper/2308.11887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11887"}},"official":{"repos":["leon1207/3dreftr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-guided-diffusion-model-for-visual","slug":"language-guided-diffusion-model-for-visual","title":"Language-Guided Diffusion Model for Visual Grounding","date":"2023-08-18","arxiv_id":"2308.09599","repositories_listed":1,"syntology":null},{"url":"/paper/3d-vista-pre-trained-transformer-for-3d","slug":"3d-vista-pre-trained-transformer-for-3d","title":"3D-VisTA: Pre-trained Transformer for 3D Vision and Text Alignment","date":"2023-08-08","arxiv_id":"2308.04352","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/3d-vista-pre-trained-transformer-for-3d#ran","syntology_url":"https://syntology.ai/paper/2308.04352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04352"}},"official":null}},{"url":"/paper/iterative-robust-visual-grounding-with-masked","slug":"iterative-robust-visual-grounding-with-masked","title":"Iterative Robust Visual Grounding with Masked Reference based Centerpoint Supervision","date":"2023-07-23","arxiv_id":"2307.12392","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-visual-grounding-with-scene-1","slug":"advancing-visual-grounding-with-scene-1","title":"Advancing Visual Grounding with Scene Knowledge: Benchmark and Method","date":"2023-07-21","arxiv_id":"2307.11558","repositories_listed":1,"syntology":null},{"url":"/paper/distilling-coarse-to-fine-semantic-matching","slug":"distilling-coarse-to-fine-semantic-matching","title":"Distilling Coarse-to-Fine Semantic Matching Knowledge for Weakly Supervised 3D Visual Grounding","date":"2023-07-18","arxiv_id":"2307.09267","repositories_listed":1,"syntology":null},{"url":"/paper/bubogpt-enabling-visual-grounding-in-multi","slug":"bubogpt-enabling-visual-grounding-in-multi","title":"BuboGPT: Enabling Visual Grounding in Multi-Modal LLMs","date":"2023-07-17","arxiv_id":"2307.08581","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":4,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bubogpt-enabling-visual-grounding-in-multi#ran","syntology_url":"https://syntology.ai/paper/2307.08581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08581"}},"official":null}},{"url":"/paper/gvcci-lifelong-learning-of-visual-grounding","slug":"gvcci-lifelong-learning-of-visual-grounding","title":"GVCCI: Lifelong Learning of Visual Grounding for Language-Guided Robotic Manipulation","date":"2023-07-12","arxiv_id":"2307.05963","repositories_listed":1,"syntology":null},{"url":"/paper/what-do-self-supervised-speech-models-know","slug":"what-do-self-supervised-speech-models-know","title":"What Do Self-Supervised Speech Models Know About Words?","date":"2023-06-30","arxiv_id":"2307.00162","repositories_listed":1,"syntology":null},{"url":"/paper/rewarded-soups-towards-pareto-optimal-1","slug":"rewarded-soups-towards-pareto-optimal-1","title":"Rewarded soups: towards Pareto-optimal alignment by interpolating weights fine-tuned on diverse rewards","date":"2023-06-07","arxiv_id":"2306.04488","repositories_listed":1,"syntology":null}],"record_sha256":"6c9ba7c1dd429ff7d651db12bab5c744ee85e43da3f24d8d69d1d770a5183cce","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}