{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-grounding/papers/ran/1","list_of":"/task/visual-grounding","task":"Visual Grounding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":111,"counts":{"archive_papers_tagged":571,"with_a_code_link":299,"where_syntology_ran_a_sample":111,"not_listed_spam_title":0,"listed":571,"listed_where_code_ran":111,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":95,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":95,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-grounding/papers/ran/1","prev":null,"next":"/task/visual-grounding/papers/ran/2","papers":[{"url":"/paper/gta1-gui-test-time-scaling-agent","slug":"gta1-gui-test-time-scaling-agent","title":"GTA1: GUI Test-time Scaling Agent","date":"2025-07-08","arxiv_id":"2507.05791","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gta1-gui-test-time-scaling-agent#ran","syntology_url":"https://syntology.ai/paper/2507.05791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.05791"}},"official":{"repos":["yan98/gta1"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cxreasonbench-a-benchmark-for-evaluating","slug":"cxreasonbench-a-benchmark-for-evaluating","title":"CXReasonBench: A Benchmark for Evaluating Structured Diagnostic Reasoning in Chest X-rays","date":"2025-05-23","arxiv_id":"2505.18087","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cxreasonbench-a-benchmark-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2505.18087","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18087"}},"official":{"repos":["ttumyche/cxreasonbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gui-g1-understanding-r1-zero-like-training","slug":"gui-g1-understanding-r1-zero-like-training","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","date":"2025-05-21","arxiv_id":"2505.15810","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/gui-g1-understanding-r1-zero-like-training#ran","syntology_url":"https://syntology.ai/paper/2505.15810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15810"}},"official":{"repos":["yuqi-zhou/gui-g1"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/deepperception-advancing-r1-like-cognitive","slug":"deepperception-advancing-r1-like-cognitive","title":"DeepPerception: Advancing R1-like Cognitive Visual Perception in MLLMs for Knowledge-Intensive Visual Grounding","date":"2025-03-17","arxiv_id":"2503.12797","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepperception-advancing-r1-like-cognitive#ran","syntology_url":"https://syntology.ai/paper/2503.12797","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.12797"}},"official":{"repos":["thunlp/deepperception"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/himtok-learning-hierarchical-mask-tokens-for","slug":"himtok-learning-hierarchical-mask-tokens-for","title":"HiMTok: Learning Hierarchical Mask Tokens for Image Segmentation with Large Multimodal Model","date":"2025-03-17","arxiv_id":"2503.13026","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":4,"n_ran_checked":4,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/himtok-learning-hierarchical-mask-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2503.13026","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13026"}},"official":{"repos":["yayafengzi/LMM-HiMTok"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/segagent-exploring-pixel-understanding-1","slug":"segagent-exploring-pixel-understanding-1","title":"SegAgent: Exploring Pixel Understanding Capabilities in MLLMs by Imitating Human Annotator Trajectories","date":"2025-03-11","arxiv_id":"2503.08625","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/segagent-exploring-pixel-understanding-1#ran","syntology_url":"https://syntology.ai/paper/2503.08625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08625"}},"official":{"repos":["aim-uofa/SegAgent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evolving-symbolic-3d-visual-grounder-with","slug":"evolving-symbolic-3d-visual-grounder-with","title":"Evolving Symbolic 3D Visual Grounder with Weakly Supervised Reflection","date":"2025-02-03","arxiv_id":"2502.01401","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/evolving-symbolic-3d-visual-grounder-with#ran","syntology_url":"https://syntology.ai/paper/2502.01401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.01401"}},"official":{"repos":["openrobotlab/ease"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/multi-task-visual-grounding-with-coarse-to","slug":"multi-task-visual-grounding-with-coarse-to","title":"Multi-task Visual Grounding with Coarse-to-Fine Consistency Constraints","date":"2025-01-12","arxiv_id":"2501.06710","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/multi-task-visual-grounding-with-coarse-to#ran","syntology_url":"https://syntology.ai/paper/2501.06710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.06710"}},"official":{"repos":["dmmm1997/c3vg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/ursa-understanding-and-verifying-chain-of","slug":"ursa-understanding-and-verifying-chain-of","title":"URSA: Understanding and Verifying Chain-of-thought Reasoning in Multimodal Mathematics","date":"2025-01-08","arxiv_id":"2501.04686","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ursa-understanding-and-verifying-chain-of#ran","syntology_url":"https://syntology.ai/paper/2501.04686","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04686"}},"official":{"repos":["URSA-MATH/URSA-MATH"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/are-vlms-ready-for-autonomous-driving-an","slug":"are-vlms-ready-for-autonomous-driving-an","title":"Are VLMs Ready for Autonomous Driving? An Empirical Study from the Reliability, Data, and Metric Perspectives","date":"2025-01-07","arxiv_id":"2501.04003","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-vlms-ready-for-autonomous-driving-an#ran","syntology_url":"https://syntology.ai/paper/2501.04003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04003"}},"official":{"repos":["opendrivelab/drivelm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-visual-grounding-a-survey","slug":"towards-visual-grounding-a-survey","title":"Towards Visual Grounding: A Survey","date":"2024-12-28","arxiv_id":"2412.20206","repositories_listed":4,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":7,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-visual-grounding-a-survey#ran","syntology_url":"https://syntology.ai/paper/2412.20206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.20206"}},"official":null}},{"url":"/paper/aria-ui-visual-grounding-for-gui-instructions","slug":"aria-ui-visual-grounding-for-gui-instructions","title":"Aria-UI: Visual Grounding for GUI Instructions","date":"2024-12-20","arxiv_id":"2412.16256","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aria-ui-visual-grounding-for-gui-instructions#ran","syntology_url":"https://syntology.ai/paper/2412.16256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.16256"}},"official":null}},{"url":"/paper/deepseek-vl2-mixture-of-experts-vision","slug":"deepseek-vl2-mixture-of-experts-vision","title":"DeepSeek-VL2: Mixture-of-Experts Vision-Language Models for Advanced Multimodal Understanding","date":"2024-12-13","arxiv_id":"2412.10302","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deepseek-vl2-mixture-of-experts-vision#ran","syntology_url":"https://syntology.ai/paper/2412.10302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.10302"}},"official":{"repos":["deepseek-ai/deepseek-vl2"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/expanding-performance-boundaries-of-open","slug":"expanding-performance-boundaries-of-open","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","date":"2024-12-06","arxiv_id":"2412.05271","repositories_listed":1,"syntology":{"n":9,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/expanding-performance-boundaries-of-open#ran","syntology_url":"https://syntology.ai/paper/2412.05271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05271"}},"official":{"repos":["opengvlab/internvl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/vlm-grounder-a-vlm-agent-for-zero-shot-3d","slug":"vlm-grounder-a-vlm-agent-for-zero-shot-3d","title":"VLM-Grounder: A VLM Agent for Zero-Shot 3D Visual Grounding","date":"2024-10-17","arxiv_id":"2410.13860","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vlm-grounder-a-vlm-agent-for-zero-shot-3d#ran","syntology_url":"https://syntology.ai/paper/2410.13860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13860"}},"official":{"repos":["openrobotlab/vlm-grounder"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/context-infused-visual-grounding-for-art","slug":"context-infused-visual-grounding-for-art","title":"Context-Infused Visual Grounding for Art","date":"2024-10-16","arxiv_id":"2410.12369","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/context-infused-visual-grounding-for-art#ran","syntology_url":"https://syntology.ai/paper/2410.12369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12369"}},"official":{"repos":["selinakhan/CIGAr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videgothink-assessing-egocentric-video","slug":"videgothink-assessing-egocentric-video","title":"VidEgoThink: Assessing Egocentric Video Understanding Capabilities for Embodied AI","date":"2024-10-15","arxiv_id":"2410.11623","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":1,"n_honours":3,"n_violates":1,"n_no_contract":9,"n_pointer_only":5,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 1 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videgothink-assessing-egocentric-video#ran","syntology_url":"https://syntology.ai/paper/2410.11623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11623"}},"official":null}},{"url":"/paper/world-to-code-multi-modal-data-generation-via","slug":"world-to-code-multi-modal-data-generation-via","title":"World to Code: Multi-modal Data Generation via Self-Instructed Compositional Captioning and Filtering","date":"2024-09-30","arxiv_id":"2409.20424","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/world-to-code-multi-modal-data-generation-via#ran","syntology_url":"https://syntology.ai/paper/2409.20424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20424"}},"official":{"repos":["foundation-multimodal-models/world2code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/simvg-a-simple-framework-for-visual-grounding","slug":"simvg-a-simple-framework-for-visual-grounding","title":"SimVG: A Simple Framework for Visual Grounding with Decoupled Multi-modal Fusion","date":"2024-09-26","arxiv_id":"2409.17531","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/simvg-a-simple-framework-for-visual-grounding#ran","syntology_url":"https://syntology.ai/paper/2409.17531","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.17531"}},"official":{"repos":["dmmm1997/simvg"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/shaking-up-vlms-comparing-transformers-and","slug":"shaking-up-vlms-comparing-transformers-and","title":"Shaking Up VLMs: Comparing Transformers and Structured State Space Models for Vision & Language Modeling","date":"2024-09-09","arxiv_id":"2409.05395","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shaking-up-vlms-comparing-transformers-and#ran","syntology_url":"https://syntology.ai/paper/2409.05395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.05395"}},"official":{"repos":["gpantaz/vl_mamba"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-grounding-with-multi-modal-conditional","slug":"visual-grounding-with-multi-modal-conditional","title":"Visual Grounding with Multi-modal Conditional Adaptation","date":"2024-09-08","arxiv_id":"2409.04999","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visual-grounding-with-multi-modal-conditional#ran","syntology_url":"https://syntology.ai/paper/2409.04999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.04999"}},"official":{"repos":["mr-bigworth/mmca"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lexicon3d-probing-visual-foundation-models","slug":"lexicon3d-probing-visual-foundation-models","title":"Lexicon3D: Probing Visual Foundation Models for Complex 3D Scene Understanding","date":"2024-09-05","arxiv_id":"2409.03757","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lexicon3d-probing-visual-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2409.03757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.03757"}},"official":null}},{"url":"/paper/iaa-inner-adaptor-architecture-empowers","slug":"iaa-inner-adaptor-architecture-empowers","title":"IAA: Inner-Adaptor Architecture Empowers Frozen Large Language Model with Multimodal Capabilities","date":"2024-08-23","arxiv_id":"2408.12902","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/iaa-inner-adaptor-architecture-empowers#ran","syntology_url":"https://syntology.ai/paper/2408.12902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12902"}},"official":{"repos":["360cvgroup/inner-adaptor-architecture"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-01942","slug":"2408-01942","title":"Visual Grounding for Object-Level Generalization in Reinforcement Learning","date":"2024-08-04","arxiv_id":"2408.01942","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/2408-01942#ran","syntology_url":"https://syntology.ai/paper/2408.01942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01942"}},"official":{"repos":["pku-rl/copl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-01120","slug":"2408-01120","title":"An Efficient and Effective Transformer Decoder-Based Framework for Multi-Task Visual Grounding","date":"2024-08-02","arxiv_id":"2408.01120","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/2408-01120#ran","syntology_url":"https://syntology.ai/paper/2408.01120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01120"}},"official":{"repos":["chenwei746/eevg"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/refmask3d-language-guided-transformer-for-3d","slug":"refmask3d-language-guided-transformer-for-3d","title":"RefMask3D: Language-Guided Transformer for 3D Referring Segmentation","date":"2024-07-25","arxiv_id":"2407.18244","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/refmask3d-language-guided-transformer-for-3d#ran","syntology_url":"https://syntology.ai/paper/2407.18244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18244"}},"official":{"repos":["heshuting555/refmask3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/3d-vision-and-language-pretraining-with-large","slug":"3d-vision-and-language-pretraining-with-large","title":"3D Vision and Language Pretraining with Large-Scale Synthetic Data","date":"2024-07-08","arxiv_id":"2407.06084","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/3d-vision-and-language-pretraining-with-large#ran","syntology_url":"https://syntology.ai/paper/2407.06084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06084"}},"official":{"repos":["idejie/3DSyn"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/smart-vision-language-reasoners","slug":"smart-vision-language-reasoners","title":"Smart Vision-Language Reasoners","date":"2024-07-05","arxiv_id":"2407.04212","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/smart-vision-language-reasoners#ran","syntology_url":"https://syntology.ai/paper/2407.04212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04212"}},"official":{"repos":["smarter-vlm/smarter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/segvg-transferring-object-bounding-box-to","slug":"segvg-transferring-object-bounding-box-to","title":"SegVG: Transferring Object Bounding Box to Segmentation for Visual Grounding","date":"2024-07-03","arxiv_id":"2407.03200","repositories_listed":1,"syntology":{"n":31,"n_ran":20,"n_constructed":15,"n_ran_checked":15,"n_instrument":5,"n_unverified":11,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":31,"phrase":"20 ran (of which 15 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 5 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/segvg-transferring-object-bounding-box-to#ran","syntology_url":"https://syntology.ai/paper/2407.03200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03200"}},"official":{"repos":["weitaikang/segvg"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":15,"n_ran_no_instrument_failure":15,"n_unverified":11,"ran_from_kinds":["official"]}}},{"url":"/paper/cambrian-1-a-fully-open-vision-centric","slug":"cambrian-1-a-fully-open-vision-centric","title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","date":"2024-06-24","arxiv_id":"2406.16860","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":3,"n_ran_checked":5,"n_instrument":5,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cambrian-1-a-fully-open-vision-centric#ran","syntology_url":"https://syntology.ai/paper/2406.16860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16860"}},"official":{"repos":["cambrian-mllm/cambrian"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vrsbench-a-versatile-vision-language","slug":"vrsbench-a-versatile-vision-language","title":"VRSBench: A Versatile Vision-Language Benchmark Dataset for Remote Sensing Image Understanding","date":"2024-06-18","arxiv_id":"2406.12384","repositories_listed":3,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":5,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vrsbench-a-versatile-vision-language#ran","syntology_url":"https://syntology.ai/paper/2406.12384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12384"}},"official":{"repos":["lx709/vrsbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/agla-mitigating-object-hallucinations-in","slug":"agla-mitigating-object-hallucinations-in","title":"AGLA: Mitigating Object Hallucinations in Large Vision-Language Models with Assembly of Global and Local Attention","date":"2024-06-18","arxiv_id":"2406.12718","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/agla-mitigating-object-hallucinations-in#ran","syntology_url":"https://syntology.ai/paper/2406.12718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12718"}},"official":{"repos":["lackel/agla"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-hierarchical-semantic-classification","slug":"learning-hierarchical-semantic-classification","title":"Visually Consistent Hierarchical Image Classification","date":"2024-06-17","arxiv_id":"2406.11608","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-hierarchical-semantic-classification#ran","syntology_url":"https://syntology.ai/paper/2406.11608","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11608"}},"official":null}},{"url":"/paper/docgenome-an-open-large-scale-scientific","slug":"docgenome-an-open-large-scale-scientific","title":"DocGenome: An Open Large-scale Scientific Document Benchmark for Training and Testing Multi-modal Large Language Models","date":"2024-06-17","arxiv_id":"2406.11633","repositories_listed":3,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/docgenome-an-open-large-scale-scientific#ran","syntology_url":"https://syntology.ai/paper/2406.11633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11633"}},"official":{"repos":["UniModal4Reasoning/DocGenome"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/separating-the-chirp-from-the-chat-self","slug":"separating-the-chirp-from-the-chat-self","title":"Separating the \"Chirp\" from the \"Chat\": Self-supervised Visual Grounding of Sound and Language","date":"2024-06-09","arxiv_id":"2406.05629","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/separating-the-chirp-from-the-chat-self#ran","syntology_url":"https://syntology.ai/paper/2406.05629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05629"}},"official":{"repos":["mhamilton723/DenseAV"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/instruction-guided-visual-masking","slug":"instruction-guided-visual-masking","title":"Instruction-Guided Visual Masking","date":"2024-05-30","arxiv_id":"2405.19783","repositories_listed":1,"syntology":{"n":28,"n_ran":20,"n_constructed":7,"n_ran_checked":11,"n_instrument":9,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"20 ran (of which 7 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 9 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/instruction-guided-visual-masking#ran","syntology_url":"https://syntology.ai/paper/2405.19783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19783"}},"official":{"repos":["2toinf/ivm"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":7,"n_ran_no_instrument_failure":11,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarial-robustness-for-visual-grounding","slug":"adversarial-robustness-for-visual-grounding","title":"Adversarial Robustness for Visual Grounding of Multimodal Large Language Models","date":"2024-05-16","arxiv_id":"2405.09981","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarial-robustness-for-visual-grounding#ran","syntology_url":"https://syntology.ai/paper/2405.09981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.09981"}},"official":{"repos":["KuofengGao/MLLM-Grounding-Robustness"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/groma-localized-visual-tokenization-for","slug":"groma-localized-visual-tokenization-for","title":"Groma: Localized Visual Tokenization for Grounding Multimodal Large Language Models","date":"2024-04-19","arxiv_id":"2404.13013","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/groma-localized-visual-tokenization-for#ran","syntology_url":"https://syntology.ai/paper/2404.13013","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13013"}},"official":{"repos":["FoundationVision/Groma"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/h2rsvlm-towards-helpful-and-honest-remote","slug":"h2rsvlm-towards-helpful-and-honest-remote","title":"VHM: Versatile and Honest Vision Language Model for Remote Sensing Image Analysis","date":"2024-03-29","arxiv_id":"2403.20213","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/h2rsvlm-towards-helpful-and-honest-remote#ran","syntology_url":"https://syntology.ai/paper/2403.20213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20213"}},"official":{"repos":["opendatalab/h2rsvlm","opendatalab/vhm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hydra-a-hyper-agent-for-dynamic-compositional","slug":"hydra-a-hyper-agent-for-dynamic-compositional","title":"HYDRA: A Hyper Agent for Dynamic Compositional Visual Reasoning","date":"2024-03-19","arxiv_id":"2403.12884","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hydra-a-hyper-agent-for-dynamic-compositional#ran","syntology_url":"https://syntology.ai/paper/2403.12884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12884"}},"official":{"repos":["ControlNet/HYDRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mikasa-multi-key-anchor-scene-aware","slug":"mikasa-multi-key-anchor-scene-aware","title":"MiKASA: Multi-Key-Anchor & Scene-Aware Transformer for 3D Visual Grounding","date":"2024-03-05","arxiv_id":"2403.03077","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":10,"n_ran_checked":11,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":16,"phrase":"11 ran (of which 10 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mikasa-multi-key-anchor-scene-aware#ran","syntology_url":"https://syntology.ai/paper/2403.03077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03077"}},"official":{"repos":["dfki-av/mikasa-3dvg"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":10,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/shapellm-universal-3d-object-understanding","slug":"shapellm-universal-3d-object-understanding","title":"ShapeLLM: Universal 3D Object Understanding for Embodied Interaction","date":"2024-02-27","arxiv_id":"2402.17766","repositories_listed":3,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":10,"n_instrument":4,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":8,"n_pointer_only":8,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/shapellm-universal-3d-object-understanding#ran","syntology_url":"https://syntology.ai/paper/2402.17766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17766"}},"official":{"repos":["qizekun/ShapeLLM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/seeing-is-believing-mitigating-hallucination","slug":"seeing-is-believing-mitigating-hallucination","title":"Seeing is Believing: Mitigating Hallucination in Large Vision-Language Models via CLIP-Guided Decoding","date":"2024-02-23","arxiv_id":"2402.15300","repositories_listed":2,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":17,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/seeing-is-believing-mitigating-hallucination#ran","syntology_url":"https://syntology.ai/paper/2402.15300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15300"}},"official":{"repos":["d-ailin/clip-guided-decoding"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-visual-and-vision-language-tracking","slug":"unifying-visual-and-vision-language-tracking","title":"Unifying Visual and Vision-Language Tracking via Contrastive Learning","date":"2024-01-20","arxiv_id":"2401.11228","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-visual-and-vision-language-tracking#ran","syntology_url":"https://syntology.ai/paper/2401.11228","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11228"}},"official":{"repos":["openspaceai/uvltrack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uncovering-the-full-potential-of-visual","slug":"uncovering-the-full-potential-of-visual","title":"Uncovering the Full Potential of Visual Grounding Methods in VQA","date":"2024-01-15","arxiv_id":"2401.07803","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uncovering-the-full-potential-of-visual#ran","syntology_url":"https://syntology.ai/paper/2401.07803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07803"}},"official":{"repos":["dreichcsl/truevg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/eyes-wide-shut-exploring-the-visual","slug":"eyes-wide-shut-exploring-the-visual","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","date":"2024-01-11","arxiv_id":"2401.06209","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eyes-wide-shut-exploring-the-visual#ran","syntology_url":"https://syntology.ai/paper/2401.06209","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06209"}},"official":{"repos":["tsb0601/MMVP"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/one-model-to-rule-them-all-towards-universal","slug":"one-model-to-rule-them-all-towards-universal","title":"One Model to Rule them All: Towards Universal Segmentation for Medical Images with Text Prompts","date":"2023-12-28","arxiv_id":"2312.17183","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/one-model-to-rule-them-all-towards-universal#ran","syntology_url":"https://syntology.ai/paper/2312.17183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.17183"}},"official":{"repos":["zhaoziheng/sat-ds"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mask-grounding-for-referring-image","slug":"mask-grounding-for-referring-image","title":"Mask Grounding for Referring Image Segmentation","date":"2023-12-19","arxiv_id":"2312.12198","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":5,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mask-grounding-for-referring-image#ran","syntology_url":"https://syntology.ai/paper/2312.12198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12198"}},"official":{"repos":["yxchng/mask-grounding"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mono3dvg-3d-visual-grounding-in-monocular","slug":"mono3dvg-3d-visual-grounding-in-monocular","title":"Mono3DVG: 3D Visual Grounding in Monocular Images","date":"2023-12-13","arxiv_id":"2312.08022","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mono3dvg-3d-visual-grounding-in-monocular#ran","syntology_url":"https://syntology.ai/paper/2312.08022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08022"}},"official":{"repos":["zhanyang-nwpu/mono3dvg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mismatch-quest-visual-and-textual-feedback","slug":"mismatch-quest-visual-and-textual-feedback","title":"Mismatch Quest: Visual and Textual Feedback for Image-Text Misalignment","date":"2023-12-05","arxiv_id":"2312.03766","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mismatch-quest-visual-and-textual-feedback#ran","syntology_url":"https://syntology.ai/paper/2312.03766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03766"}},"official":{"repos":["mismatchquest/mismatchquest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aligning-and-prompting-everything-all-at-once","slug":"aligning-and-prompting-everything-all-at-once","title":"Aligning and Prompting Everything All at Once for Universal Visual Perception","date":"2023-12-04","arxiv_id":"2312.02153","repositories_listed":2,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/aligning-and-prompting-everything-all-at-once#ran","syntology_url":"https://syntology.ai/paper/2312.02153","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02153"}},"official":{"repos":["shenyunhang/ape"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/g2d-from-global-to-dense-radiography","slug":"g2d-from-global-to-dense-radiography","title":"G2D: From Global to Dense Radiography Representation Learning via Vision-Language Pre-training","date":"2023-12-03","arxiv_id":"2312.01522","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/g2d-from-global-to-dense-radiography#ran","syntology_url":"https://syntology.ai/paper/2312.01522","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01522"}},"official":{"repos":["cheliu-computation/g2d-neurips24"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/behind-the-magic-merlim-multi-modal","slug":"behind-the-magic-merlim-multi-modal","title":"Behind the Magic, MERLIM: Multi-modal Evaluation Benchmark for Large Image-Language Models","date":"2023-12-03","arxiv_id":"2312.02219","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/behind-the-magic-merlim-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2312.02219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02219"}},"official":{"repos":["ojedaf/merlim"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-referring-expression-comprehension","slug":"zero-shot-referring-expression-comprehension","title":"Zero-shot Referring Expression Comprehension via Structural Similarity Between Images and Captions","date":"2023-11-28","arxiv_id":"2311.17048","repositories_listed":1,"syntology":{"n":20,"n_ran":14,"n_constructed":0,"n_ran_checked":10,"n_instrument":4,"n_unverified":6,"n_honours":0,"n_violates":3,"n_no_contract":7,"n_pointer_only":10,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 3 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/zero-shot-referring-expression-comprehension#ran","syntology_url":"https://syntology.ai/paper/2311.17048","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17048"}},"official":{"repos":["show-han/zeroshot_rec"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-programming-for-zero-shot-open","slug":"visual-programming-for-zero-shot-open","title":"Visual Programming for Zero-shot Open-Vocabulary 3D Visual Grounding","date":"2023-11-26","arxiv_id":"2311.15383","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-programming-for-zero-shot-open#ran","syntology_url":"https://syntology.ai/paper/2311.15383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15383"}},"official":{"repos":["CurryYuan/ZSVG3D"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/next-chat-an-lmm-for-chat-detection-and","slug":"next-chat-an-lmm-for-chat-detection-and","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","date":"2023-11-08","arxiv_id":"2311.04498","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/next-chat-an-lmm-for-chat-detection-and#ran","syntology_url":"https://syntology.ai/paper/2311.04498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04498"}},"official":{"repos":["next-chatv/next-chat"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cityrefer-geography-aware-3d-visual-grounding","slug":"cityrefer-geography-aware-3d-visual-grounding","title":"CityRefer: Geography-aware 3D Visual Grounding Dataset on City-scale Point Cloud Data","date":"2023-10-28","arxiv_id":"2310.18773","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cityrefer-geography-aware-3d-visual-grounding#ran","syntology_url":"https://syntology.ai/paper/2310.18773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18773"}},"official":{"repos":["atr-dbi/cityrefer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/groovist-a-metric-for-grounding-objects-in","slug":"groovist-a-metric-for-grounding-objects-in","title":"GROOViST: A Metric for Grounding Objects in Visual Storytelling","date":"2023-10-26","arxiv_id":"2310.17770","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/groovist-a-metric-for-grounding-objects-in#ran","syntology_url":"https://syntology.ai/paper/2310.17770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17770"}},"official":{"repos":["akskuchi/groovist"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-grounding-helps-learn-word-meanings-in","slug":"visual-grounding-helps-learn-word-meanings-in","title":"Visual Grounding Helps Learn Word Meanings in Low-Data Regimes","date":"2023-10-20","arxiv_id":"2310.13257","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-grounding-helps-learn-word-meanings-in#ran","syntology_url":"https://syntology.ai/paper/2310.13257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13257"}},"official":{"repos":["EvLab-MIT/LexiContrastiveGrd"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/minigpt-v2-large-language-model-as-a-unified","slug":"minigpt-v2-large-language-model-as-a-unified","title":"MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning","date":"2023-10-14","arxiv_id":"2310.09478","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minigpt-v2-large-language-model-as-a-unified#ran","syntology_url":"https://syntology.ai/paper/2310.09478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09478"}},"official":null}},{"url":"/paper/cot3dref-chain-of-thoughts-data-efficient-3d","slug":"cot3dref-chain-of-thoughts-data-efficient-3d","title":"CoT3DRef: Chain-of-Thoughts Data-Efficient 3D Visual Grounding","date":"2023-10-10","arxiv_id":"2310.06214","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cot3dref-chain-of-thoughts-data-efficient-3d#ran","syntology_url":"https://syntology.ai/paper/2310.06214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06214"}},"official":{"repos":["eslambakr/CoT3D_VG"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rephrase-augment-reason-visual-grounding-of","slug":"rephrase-augment-reason-visual-grounding-of","title":"Rephrase, Augment, Reason: Visual Grounding of Questions for Vision-Language Models","date":"2023-10-09","arxiv_id":"2310.05861","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rephrase-augment-reason-visual-grounding-of#ran","syntology_url":"https://syntology.ai/paper/2310.05861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05861"}},"official":{"repos":["archiki/repare"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-grounder-open-vocabulary-3d-visual","slug":"llm-grounder-open-vocabulary-3d-visual","title":"LLM-Grounder: Open-Vocabulary 3D Visual Grounding with Large Language Model as an Agent","date":"2023-09-21","arxiv_id":"2309.12311","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llm-grounder-open-vocabulary-3d-visual#ran","syntology_url":"https://syntology.ai/paper/2309.12311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12311"}},"official":{"repos":["sled-group/chat-with-nerf"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi3drefer-grounding-text-description-to","slug":"multi3drefer-grounding-text-description-to","title":"Multi3DRefer: Grounding Text Description to Multiple 3D Objects","date":"2023-09-11","arxiv_id":"2309.05251","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multi3drefer-grounding-text-description-to#ran","syntology_url":"https://syntology.ai/paper/2309.05251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05251"}},"official":{"repos":["3dlg-hcvc/M3DRef-CLIP"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-unified-framework-for-3d-point-cloud-visual","slug":"a-unified-framework-for-3d-point-cloud-visual","title":"A Unified Framework for 3D Point Cloud Visual Grounding","date":"2023-08-23","arxiv_id":"2308.11887","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-framework-for-3d-point-cloud-visual#ran","syntology_url":"https://syntology.ai/paper/2308.11887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11887"}},"official":{"repos":["leon1207/3dreftr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/3d-vista-pre-trained-transformer-for-3d","slug":"3d-vista-pre-trained-transformer-for-3d","title":"3D-VisTA: Pre-trained Transformer for 3D Vision and Text Alignment","date":"2023-08-08","arxiv_id":"2308.04352","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/3d-vista-pre-trained-transformer-for-3d#ran","syntology_url":"https://syntology.ai/paper/2308.04352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04352"}},"official":null}},{"url":"/paper/bubogpt-enabling-visual-grounding-in-multi","slug":"bubogpt-enabling-visual-grounding-in-multi","title":"BuboGPT: Enabling Visual Grounding in Multi-Modal LLMs","date":"2023-07-17","arxiv_id":"2307.08581","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":4,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bubogpt-enabling-visual-grounding-in-multi#ran","syntology_url":"https://syntology.ai/paper/2307.08581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08581"}},"official":null}},{"url":"/paper/an-examination-of-the-robustness-of-reference","slug":"an-examination-of-the-robustness-of-reference","title":"An Examination of the Robustness of Reference-Free Image Captioning Evaluation Metrics","date":"2023-05-24","arxiv_id":"2305.14998","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-examination-of-the-robustness-of-reference#ran","syntology_url":"https://syntology.ai/paper/2305.14998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14998"}},"official":{"repos":["saba96/img-cap-metrics-robustness"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/syllable-discovery-and-cross-lingual","slug":"syllable-discovery-and-cross-lingual","title":"Syllable Discovery and Cross-Lingual Generalization in a Visually Grounded, Self-Supervised Speech Model","date":"2023-05-19","arxiv_id":"2305.11435","repositories_listed":2,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 2 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/syllable-discovery-and-cross-lingual#ran","syntology_url":"https://syntology.ai/paper/2305.11435","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11435"}},"official":{"repos":["jasonppy/syllable-discovery"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/viewrefer-grasp-the-multi-view-knowledge-for","slug":"viewrefer-grasp-the-multi-view-knowledge-for","title":"ViewRefer: Grasp the Multi-view Knowledge for 3D Visual Grounding with GPT and Prototype Guidance","date":"2023-03-29","arxiv_id":"2303.16894","repositories_listed":7,"syntology":{"n":12,"n_ran":8,"n_constructed":2,"n_ran_checked":6,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":12,"phrase":"8 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/viewrefer-grasp-the-multi-view-knowledge-for#ran","syntology_url":"https://syntology.ai/paper/2303.16894","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16894"}},"official":{"repos":["ivan-tang-3d/viewrefer3d","ziyuguo99/viewrefer3d"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/joint-visual-grounding-and-tracking-with","slug":"joint-visual-grounding-and-tracking-with","title":"Joint Visual Grounding and Tracking with Natural Language Specification","date":"2023-03-21","arxiv_id":"2303.12027","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/joint-visual-grounding-and-tracking-with#ran","syntology_url":"https://syntology.ai/paper/2303.12027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.12027"}},"official":{"repos":["lizhou-cs/jointnlt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mplug-2-a-modularized-multi-modal-foundation","slug":"mplug-2-a-modularized-multi-modal-foundation","title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","date":"2023-02-01","arxiv_id":"2302.00402","repositories_listed":4,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":9,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mplug-2-a-modularized-multi-modal-foundation#ran","syntology_url":"https://syntology.ai/paper/2302.00402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00402"}},"official":{"repos":["alibaba/AliceMind"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/position-guided-text-prompt-for-vision","slug":"position-guided-text-prompt-for-vision","title":"Position-guided Text Prompt for Vision-Language Pre-training","date":"2022-12-19","arxiv_id":"2212.09737","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/position-guided-text-prompt-for-vision#ran","syntology_url":"https://syntology.ai/paper/2212.09737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09737"}},"official":{"repos":["sail-sg/ptp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/look-around-and-refer-2d-synthetic-semantics","slug":"look-around-and-refer-2d-synthetic-semantics","title":"Look Around and Refer: 2D Synthetic Semantics Knowledge Distillation for 3D Visual Grounding","date":"2022-11-25","arxiv_id":"2211.14241","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/look-around-and-refer-2d-synthetic-semantics#ran","syntology_url":"https://syntology.ai/paper/2211.14241","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.14241"}},"official":{"repos":["eslambakr/LAR-Look-Around-and-Refer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","slug":"x-2-vlm-all-in-one-pre-trained-model-for","title":"X$^2$-VLM: All-In-One Pre-trained Model For Vision-Language Tasks","date":"2022-11-22","arxiv_id":"2211.12402","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/x-2-vlm-all-in-one-pre-trained-model-for#ran","syntology_url":"https://syntology.ai/paper/2211.12402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12402"}},"official":{"repos":["zengyan-97/x2-vlm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/instruction-following-agents-with-jointly-pre","slug":"instruction-following-agents-with-jointly-pre","title":"Instruction-Following Agents with Multimodal Transformer","date":"2022-10-24","arxiv_id":"2210.13431","repositories_listed":1,"syntology":{"n":20,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/instruction-following-agents-with-jointly-pre#ran","syntology_url":"https://syntology.ai/paper/2210.13431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13431"}},"official":{"repos":["lhao499/instructrl"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/ham-hierarchical-attention-model-with-high","slug":"ham-hierarchical-attention-model-with-high","title":"Learning Point-Language Hierarchical Alignment for 3D Visual Grounding","date":"2022-10-22","arxiv_id":"2210.12513","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ham-hierarchical-attention-model-with-high#ran","syntology_url":"https://syntology.ai/paper/2210.12513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.12513"}},"official":{"repos":["ppjmchen/ham"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/eda-explicit-text-decoupling-and-dense","slug":"eda-explicit-text-decoupling-and-dense","title":"EDA: Explicit Text-Decoupling and Dense Alignment for 3D Visual Grounding","date":"2022-09-29","arxiv_id":"2209.14941","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eda-explicit-text-decoupling-and-dense#ran","syntology_url":"https://syntology.ai/paper/2209.14941","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.14941"}},"official":{"repos":["yanmin-wu/eda"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/introspective-learning-a-two-stage-approach-1","slug":"introspective-learning-a-two-stage-approach-1","title":"Introspective Learning : A Two-Stage Approach for Inference in Neural Networks","date":"2022-09-17","arxiv_id":"2209.08425","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/introspective-learning-a-two-stage-approach-1#ran","syntology_url":"https://syntology.ai/paper/2209.08425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.08425"}},"official":{"repos":["olivesgatech/introspective-learning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/efficient-vision-language-pretraining-with","slug":"efficient-vision-language-pretraining-with","title":"Efficient Vision-Language Pretraining with Visual Concepts and Hierarchical Alignment","date":"2022-08-29","arxiv_id":"2208.13628","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-vision-language-pretraining-with#ran","syntology_url":"https://syntology.ai/paper/2208.13628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.13628"}},"official":{"repos":["mshukor/vicha"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/siri-a-simple-selective-retraining-mechanism","slug":"siri-a-simple-selective-retraining-mechanism","title":"SiRi: A Simple Selective Retraining Mechanism for Transformer-based Visual Grounding","date":"2022-07-27","arxiv_id":"2207.13325","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":2,"n_no_contract":3,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/siri-a-simple-selective-retraining-mechanism#ran","syntology_url":"https://syntology.ai/paper/2207.13325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.13325"}},"official":{"repos":["qumengxue/siri-vg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/3d-sps-single-stage-3d-visual-grounding-via","slug":"3d-sps-single-stage-3d-visual-grounding-via","title":"3D-SPS: Single-Stage 3D Visual Grounding via Referred Point Progressive Selection","date":"2022-04-13","arxiv_id":"2204.06272","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/3d-sps-single-stage-3d-visual-grounding-via#ran","syntology_url":"https://syntology.ai/paper/2204.06272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.06272"}},"official":{"repos":["fjhzhixi/3d-sps"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-view-transformer-for-3d-visual","slug":"multi-view-transformer-for-3d-visual","title":"Multi-View Transformer for 3D Visual Grounding","date":"2022-04-05","arxiv_id":"2204.02174","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-view-transformer-for-3d-visual#ran","syntology_url":"https://syntology.ai/paper/2204.02174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02174"}},"official":{"repos":["sega-hsj/mvt-3dvg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/seqtr-a-simple-yet-universal-network-for","slug":"seqtr-a-simple-yet-universal-network-for","title":"SeqTR: A Simple yet Universal Network for Visual Grounding","date":"2022-03-30","arxiv_id":"2203.16265","repositories_listed":3,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seqtr-a-simple-yet-universal-network-for#ran","syntology_url":"https://syntology.ai/paper/2203.16265","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16265"}},"official":{"repos":["sean-zhuh/seqtr"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/tubedetr-spatio-temporal-video-grounding-with","slug":"tubedetr-spatio-temporal-video-grounding-with","title":"TubeDETR: Spatio-Temporal Video Grounding with Transformers","date":"2022-03-30","arxiv_id":"2203.16434","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/tubedetr-spatio-temporal-video-grounding-with#ran","syntology_url":"https://syntology.ai/paper/2203.16434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16434"}},"official":{"repos":["antoyang/TubeDETR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/collaborative-transformers-for-grounded","slug":"collaborative-transformers-for-grounded","title":"Collaborative Transformers for Grounded Situation Recognition","date":"2022-03-30","arxiv_id":"2203.16518","repositories_listed":3,"syntology":{"n":7,"n_ran":5,"n_constructed":3,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/collaborative-transformers-for-grounded#ran","syntology_url":"https://syntology.ai/paper/2203.16518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16518"}},"official":{"repos":["jhcho99/coformer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/local-global-context-aware-transformer-for","slug":"local-global-context-aware-transformer-for","title":"Local-Global Context Aware Transformer for Language-Guided Video Segmentation","date":"2022-03-18","arxiv_id":"2203.09773","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/local-global-context-aware-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2203.09773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09773"}},"official":{"repos":["leonnnop/locater"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pseudo-q-generating-pseudo-language-queries","slug":"pseudo-q-generating-pseudo-language-queries","title":"Pseudo-Q: Generating Pseudo Language Queries for Visual Grounding","date":"2022-03-16","arxiv_id":"2203.08481","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":7,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pseudo-q-generating-pseudo-language-queries#ran","syntology_url":"https://syntology.ai/paper/2203.08481","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08481"}},"official":{"repos":["leaplabthu/pseudo-q"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rex-reasoning-aware-and-grounded-explanation","slug":"rex-reasoning-aware-and-grounded-explanation","title":"REX: Reasoning-aware and Grounded Explanation","date":"2022-03-11","arxiv_id":"2203.06107","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rex-reasoning-aware-and-grounded-explanation#ran","syntology_url":"https://syntology.ai/paper/2203.06107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.06107"}},"official":{"repos":["szzexpoi/rex"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-architectures-tasks-and-modalities","slug":"unifying-architectures-tasks-and-modalities","title":"OFA: Unifying Architectures, Tasks, and Modalities Through a Simple Sequence-to-Sequence Learning Framework","date":"2022-02-07","arxiv_id":"2202.03052","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-architectures-tasks-and-modalities#ran","syntology_url":"https://syntology.ai/paper/2202.03052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03052"}},"official":{"repos":["ofa-sys/ofa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/clip-lite-information-efficient-visual","slug":"clip-lite-information-efficient-visual","title":"CLIP-Lite: Information Efficient Visual Representation Learning with Language Supervision","date":"2021-12-14","arxiv_id":"2112.07133","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/clip-lite-information-efficient-visual#ran","syntology_url":"https://syntology.ai/paper/2112.07133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.07133"}},"official":{"repos":["4m4n5/CLIP-Lite"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/crossing-the-format-boundary-of-text-and","slug":"crossing-the-format-boundary-of-text-and","title":"UniTAB: Unifying Text and Box Outputs for Grounded Vision-Language Modeling","date":"2021-11-23","arxiv_id":"2111.12085","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":9,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/crossing-the-format-boundary-of-text-and#ran","syntology_url":"https://syntology.ai/paper/2111.12085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12085"}},"official":{"repos":["microsoft/UniTAB"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/grounded-situation-recognition-with","slug":"grounded-situation-recognition-with","title":"Grounded Situation Recognition with Transformers","date":"2021-11-19","arxiv_id":"2111.10135","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/grounded-situation-recognition-with#ran","syntology_url":"https://syntology.ai/paper/2111.10135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.10135"}},"official":{"repos":["jhcho99/gsrtr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-grained-vision-language-pre-training","slug":"multi-grained-vision-language-pre-training","title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","date":"2021-11-16","arxiv_id":"2111.08276","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-grained-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2111.08276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.08276"}},"official":{"repos":["zengyan-97/x-vlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cpt-colorful-prompt-tuning-for-pre-trained","slug":"cpt-colorful-prompt-tuning-for-pre-trained","title":"CPT: Colorful Prompt Tuning for Pre-trained Vision-Language Models","date":"2021-09-24","arxiv_id":"2109.11797","repositories_listed":2,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cpt-colorful-prompt-tuning-for-pre-trained#ran","syntology_url":"https://syntology.ai/paper/2109.11797","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.11797"}},"official":{"repos":["thunlp/cpt"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/referring-transformer-a-one-step-approach-to","slug":"referring-transformer-a-one-step-approach-to","title":"Referring Transformer: A One-step Approach to Multi-task Visual Grounding","date":"2021-06-06","arxiv_id":"2106.03089","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/referring-transformer-a-one-step-approach-to#ran","syntology_url":"https://syntology.ai/paper/2106.03089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03089"}},"official":{"repos":["ubc-vision/RefTR"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mdetr-modulated-detection-for-end-to-end","slug":"mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","arxiv_id":"2104.12763","repositories_listed":5,"syntology":{"n":11,"n_ran":7,"n_constructed":4,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mdetr-modulated-detection-for-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2104.12763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.12763"}},"official":{"repos":["ashkamath/mdetr"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/transvg-end-to-end-visual-grounding-with","slug":"transvg-end-to-end-visual-grounding-with","title":"TransVG: End-to-End Visual Grounding with Transformers","date":"2021-04-17","arxiv_id":"2104.08541","repositories_listed":2,"syntology":{"n":20,"n_ran":18,"n_constructed":0,"n_ran_checked":13,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":8,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/transvg-end-to-end-visual-grounding-with#ran","syntology_url":"https://syntology.ai/paper/2104.08541","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08541"}},"official":{"repos":["djiajunustc/TransVG"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/cyclic-co-learning-of-sounding-object-visual","slug":"cyclic-co-learning-of-sounding-object-visual","title":"Cyclic Co-Learning of Sounding Object Visual Grounding and Sound Separation","date":"2021-04-05","arxiv_id":"2104.02026","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cyclic-co-learning-of-sounding-object-visual#ran","syntology_url":"https://syntology.ai/paper/2104.02026","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.02026"}},"official":{"repos":["YapengTian/CCOL-CVPR21"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/relation-aware-instance-refinement-for-weakly","slug":"relation-aware-instance-refinement-for-weakly","title":"Relation-aware Instance Refinement for Weakly Supervised Visual Grounding","date":"2021-03-24","arxiv_id":"2103.12989","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/relation-aware-instance-refinement-for-weakly#ran","syntology_url":"https://syntology.ai/paper/2103.12989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.12989"}},"official":{"repos":["youngfly11/ReIR-WeaklyGrounding.pytorch"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"6c1166b5992e025fbc9351c0a51ec412a87f7176b52b2726738e26d93aab176b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}