{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-reasoning/papers/ran/1","list_of":"/task/visual-reasoning","task":"Visual Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":165,"counts":{"archive_papers_tagged":698,"with_a_code_link":356,"where_syntology_ran_a_sample":165,"not_listed_spam_title":0,"listed":698,"listed_where_code_ran":165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":130,"every_run_a_failure_of_syntologys_instrument":35,"listed_with_a_run_with_no_instrument_failure":130,"listed_every_run_a_failure_of_syntologys_instrument":35,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-reasoning/papers/ran/1","prev":null,"next":"/task/visual-reasoning/papers/ran/2","papers":[{"url":"/paper/skywork-r1v3-technical-report","slug":"skywork-r1v3-technical-report","title":"Skywork-R1V3 Technical Report","date":"2025-07-08","arxiv_id":"2507.06167","repositories_listed":2,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/skywork-r1v3-technical-report#ran","syntology_url":"https://syntology.ai/paper/2507.06167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.06167"}},"official":{"repos":["SkyworkAI/Skywork-R1V"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/machine-mental-imagery-empower-multimodal","slug":"machine-mental-imagery-empower-multimodal","title":"Machine Mental Imagery: Empower Multimodal Reasoning with Latent Visual Tokens","date":"2025-06-20","arxiv_id":"2506.17218","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":17,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/machine-mental-imagery-empower-multimodal#ran","syntology_url":"https://syntology.ai/paper/2506.17218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.17218"}},"official":{"repos":["umass-embodied-agi/mirage"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vrest-enhancing-reasoning-in-large-vision","slug":"vrest-enhancing-reasoning-in-large-vision","title":"VReST: Enhancing Reasoning in Large Vision-Language Models through Tree Search and Self-Reward Mechanism","date":"2025-06-10","arxiv_id":"2506.08691","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":4,"n_instrument":8,"n_unverified":3,"n_honours":3,"n_violates":1,"n_no_contract":0,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 1 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vrest-enhancing-reasoning-in-large-vision#ran","syntology_url":"https://syntology.ai/paper/2506.08691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08691"}},"official":{"repos":["garyjiajia/vrest"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/understand-think-and-answer-advancing-visual","slug":"understand-think-and-answer-advancing-visual","title":"Understand, Think, and Answer: Advancing Visual Reasoning with Large Multimodal Models","date":"2025-05-27","arxiv_id":"2505.20753","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/understand-think-and-answer-advancing-visual#ran","syntology_url":"https://syntology.ai/paper/2505.20753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20753"}},"official":{"repos":["jefferyzhan/griffon"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/viscra-a-visual-chain-reasoning-attack-for","slug":"viscra-a-visual-chain-reasoning-attack-for","title":"VisCRA: A Visual Chain Reasoning Attack for Jailbreaking Multimodal Large Language Models","date":"2025-05-26","arxiv_id":"2505.19684","repositories_listed":0,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/viscra-a-visual-chain-reasoning-attack-for#ran","syntology_url":"https://syntology.ai/paper/2505.19684","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19684"}},"official":null}},{"url":"/paper/are-vision-language-models-ready-for-clinical","slug":"are-vision-language-models-ready-for-clinical","title":"Are Vision Language Models Ready for Clinical Diagnosis? A 3D Medical Benchmark for Tumor-centric Visual Question Answering","date":"2025-05-25","arxiv_id":"2505.18915","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-vision-language-models-ready-for-clinical#ran","syntology_url":"https://syntology.ai/paper/2505.18915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18915"}},"official":{"repos":["schuture/deeptumorvqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vtool-r1-vlms-learn-to-think-with-images-via","slug":"vtool-r1-vlms-learn-to-think-with-images-via","title":"VTool-R1: VLMs Learn to Think with Images via Reinforcement Learning on Multimodal Tool Use","date":"2025-05-25","arxiv_id":"2505.19255","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vtool-r1-vlms-learn-to-think-with-images-via#ran","syntology_url":"https://syntology.ai/paper/2505.19255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19255"}},"official":null}},{"url":"/paper/futuresightdrive-thinking-visually-with","slug":"futuresightdrive-thinking-visually-with","title":"FutureSightDrive: Thinking Visually with Spatio-Temporal CoT for Autonomous Driving","date":"2025-05-23","arxiv_id":"2505.17685","repositories_listed":0,"syntology":{"n":9,"n_ran":6,"n_constructed":5,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"6 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/futuresightdrive-thinking-visually-with#ran","syntology_url":"https://syntology.ai/paper/2505.17685","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17685"}},"official":null}},{"url":"/paper/daily-omni-towards-audio-visual-reasoning","slug":"daily-omni-towards-audio-visual-reasoning","title":"Daily-Omni: Towards Audio-Visual Reasoning with Temporal Alignment across Modalities","date":"2025-05-23","arxiv_id":"2505.17862","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/daily-omni-towards-audio-visual-reasoning#ran","syntology_url":"https://syntology.ai/paper/2505.17862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17862"}},"official":{"repos":["lliar-liar/daily-omni"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/danmakutppbench-a-multi-modal-benchmark-for","slug":"danmakutppbench-a-multi-modal-benchmark-for","title":"DanmakuTPPBench: A Multi-modal Benchmark for Temporal Point Process Modeling and Understanding","date":"2025-05-23","arxiv_id":"2505.18411","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/danmakutppbench-a-multi-modal-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2505.18411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18411"}},"official":{"repos":["frenkie-chiang/danmakutppbench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/lavida-a-large-diffusion-language-model-for","slug":"lavida-a-large-diffusion-language-model-for","title":"LaViDa: A Large Diffusion Language Model for Multimodal Understanding","date":"2025-05-22","arxiv_id":"2505.16839","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/lavida-a-large-diffusion-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2505.16839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16839"}},"official":{"repos":["jacklishufan/lavida"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/ocr-reasoning-benchmark-unveiling-the-true","slug":"ocr-reasoning-benchmark-unveiling-the-true","title":"OCR-Reasoning Benchmark: Unveiling the True Capabilities of MLLMs in Complex Text-Rich Image Reasoning","date":"2025-05-22","arxiv_id":"2505.17163","repositories_listed":0,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ocr-reasoning-benchmark-unveiling-the-true#ran","syntology_url":"https://syntology.ai/paper/2505.17163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17163"}},"official":null}},{"url":"/paper/deepeyes-incentivizing-thinking-with-images","slug":"deepeyes-incentivizing-thinking-with-images","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14362","repositories_listed":1,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deepeyes-incentivizing-thinking-with-images#ran","syntology_url":"https://syntology.ai/paper/2505.14362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14362"}},"official":{"repos":["visual-agent/deepeyes"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/visualquality-r1-reasoning-induced-image","slug":"visualquality-r1-reasoning-induced-image","title":"VisualQuality-R1: Reasoning-Induced Image Quality Assessment via Reinforcement Learning to Rank","date":"2025-05-20","arxiv_id":"2505.14460","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/visualquality-r1-reasoning-induced-image#ran","syntology_url":"https://syntology.ai/paper/2505.14460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14460"}},"official":{"repos":["tianhewu/visualquality-r1"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/doxing-via-the-lens-revealing-privacy-leakage","slug":"doxing-via-the-lens-revealing-privacy-leakage","title":"Doxing via the Lens: Revealing Location-related Privacy Leakage on Multi-modal Large Reasoning Models","date":"2025-04-27","arxiv_id":"2504.19373","repositories_listed":0,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/doxing-via-the-lens-revealing-privacy-leakage#ran","syntology_url":"https://syntology.ai/paper/2504.19373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.19373"}},"official":null}},{"url":"/paper/noisyrollout-reinforcing-visual-reasoning","slug":"noisyrollout-reinforcing-visual-reasoning","title":"NoisyRollout: Reinforcing Visual Reasoning with Data Augmentation","date":"2025-04-17","arxiv_id":"2504.13055","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/noisyrollout-reinforcing-visual-reasoning#ran","syntology_url":"https://syntology.ai/paper/2504.13055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13055"}},"official":null}},{"url":"/paper/sota-with-less-mcts-guided-sample-selection","slug":"sota-with-less-mcts-guided-sample-selection","title":"SoTA with Less: MCTS-Guided Sample Selection for Data-Efficient Visual Reasoning Self-Improvement","date":"2025-04-10","arxiv_id":"2504.07934","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sota-with-less-mcts-guided-sample-selection#ran","syntology_url":"https://syntology.ai/paper/2504.07934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07934"}},"official":{"repos":["si0wang/thinklite-vl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/reason-rft-reinforcement-fine-tuning-for","slug":"reason-rft-reinforcement-fine-tuning-for","title":"Reason-RFT: Reinforcement Fine-Tuning for Visual Reasoning","date":"2025-03-26","arxiv_id":"2503.20752","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reason-rft-reinforcement-fine-tuning-for#ran","syntology_url":"https://syntology.ai/paper/2503.20752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20752"}},"official":null}},{"url":"/paper/agentic-keyframe-search-for-video-question","slug":"agentic-keyframe-search-for-video-question","title":"Agentic Keyframe Search for Video Question Answering","date":"2025-03-20","arxiv_id":"2503.16032","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/agentic-keyframe-search-for-video-question#ran","syntology_url":"https://syntology.ai/paper/2503.16032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16032"}},"official":{"repos":["fansunqi/akeys"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/segagent-exploring-pixel-understanding-1","slug":"segagent-exploring-pixel-understanding-1","title":"SegAgent: Exploring Pixel Understanding Capabilities in MLLMs by Imitating Human Annotator Trajectories","date":"2025-03-11","arxiv_id":"2503.08625","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/segagent-exploring-pixel-understanding-1#ran","syntology_url":"https://syntology.ai/paper/2503.08625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08625"}},"official":{"repos":["aim-uofa/SegAgent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/r1-zero-s-aha-moment-in-visual-reasoning-on-a","slug":"r1-zero-s-aha-moment-in-visual-reasoning-on-a","title":"R1-Zero's \"Aha Moment\" in Visual Reasoning on a 2B Non-SFT Model","date":"2025-03-07","arxiv_id":"2503.05132","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r1-zero-s-aha-moment-in-visual-reasoning-on-a#ran","syntology_url":"https://syntology.ai/paper/2503.05132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05132"}},"official":{"repos":["turningpoint-ai/visualthinker-r1-zero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/forgotten-polygons-multimodal-large-language","slug":"forgotten-polygons-multimodal-large-language","title":"Forgotten Polygons: Multimodal Large Language Models are Shape-Blind","date":"2025-02-21","arxiv_id":"2502.15969","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/forgotten-polygons-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2502.15969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15969"}},"official":{"repos":["rsinghlab/shape-blind"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/enhancing-cognition-and-explainability-of","slug":"enhancing-cognition-and-explainability-of","title":"Enhancing Cognition and Explainability of Multimodal Foundation Models with Self-Synthesized Data","date":"2025-02-19","arxiv_id":"2502.14044","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-cognition-and-explainability-of#ran","syntology_url":"https://syntology.ai/paper/2502.14044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14044"}},"official":{"repos":["sycny/selfsynthx"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/cityeqa-a-hierarchical-llm-agent-on-embodied","slug":"cityeqa-a-hierarchical-llm-agent-on-embodied","title":"CityEQA: A Hierarchical LLM Agent on Embodied Question Answering Benchmark in City Space","date":"2025-02-18","arxiv_id":"2502.12532","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/cityeqa-a-hierarchical-llm-agent-on-embodied#ran","syntology_url":"https://syntology.ai/paper/2502.12532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.12532"}},"official":{"repos":["biluyong/cityeqa","tsinghua-fib-lab/CityEQA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/generalizing-from-simple-to-hard-visual","slug":"generalizing-from-simple-to-hard-visual","title":"Generalizing from SIMPLE to HARD Visual Reasoning: Can We Mitigate Modality Imbalance in VLMs?","date":"2025-01-05","arxiv_id":"2501.02669","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizing-from-simple-to-hard-visual#ran","syntology_url":"https://syntology.ai/paper/2501.02669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.02669"}},"official":{"repos":["princeton-pli/vlm_s2h"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-visual-abstract-reasoning-through","slug":"learning-visual-abstract-reasoning-through","title":"Learning Visual Abstract Reasoning through Dual-Stream Networks","date":"2024-11-29","arxiv_id":"2411.19451","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":8,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 8 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 8 samples that ran constructed an object rather than computing a result","sample_list":"/paper/learning-visual-abstract-reasoning-through#ran","syntology_url":"https://syntology.ai/paper/2411.19451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19451"}},"official":{"repos":["VecchioID/DRNet"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":8,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/immune-improving-safety-against-jailbreaks-in","slug":"immune-improving-safety-against-jailbreaks-in","title":"Immune: Improving Safety Against Jailbreaks in Multi-modal LLMs via Inference-Time Alignment","date":"2024-11-27","arxiv_id":"2411.18688","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/immune-improving-safety-against-jailbreaks-in#ran","syntology_url":"https://syntology.ai/paper/2411.18688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18688"}},"official":null}},{"url":"/paper/insight-v-exploring-long-chain-visual","slug":"insight-v-exploring-long-chain-visual","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","date":"2024-11-21","arxiv_id":"2411.14432","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/insight-v-exploring-long-chain-visual#ran","syntology_url":"https://syntology.ai/paper/2411.14432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14432"}},"official":{"repos":["dongyh20/insight-v"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/clevrskills-compositional-language-and-visual","slug":"clevrskills-compositional-language-and-visual","title":"ClevrSkills: Compositional Language and Visual Reasoning in Robotics","date":"2024-11-13","arxiv_id":"2411.09052","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/clevrskills-compositional-language-and-visual#ran","syntology_url":"https://syntology.ai/paper/2411.09052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.09052"}},"official":{"repos":["Qualcomm-AI-research/ClevrSkills"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hourvideo-1-hour-video-language-understanding","slug":"hourvideo-1-hour-video-language-understanding","title":"HourVideo: 1-Hour Video-Language Understanding","date":"2024-11-07","arxiv_id":"2411.04998","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hourvideo-1-hour-video-language-understanding#ran","syntology_url":"https://syntology.ai/paper/2411.04998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04998"}},"official":{"repos":["keshik6/HourVideo"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/on-erroneous-agreements-of-clip-image","slug":"on-erroneous-agreements-of-clip-image","title":"On Erroneous Agreements of CLIP Image Embeddings","date":"2024-11-07","arxiv_id":"2411.05195","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/on-erroneous-agreements-of-clip-image#ran","syntology_url":"https://syntology.ai/paper/2411.05195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.05195"}},"official":{"repos":["lst627/CLIP-Embeds"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/inference-optimal-vlms-need-only-one-visual","slug":"inference-optimal-vlms-need-only-one-visual","title":"Inference Optimal VLMs Need Fewer Visual Tokens and More Parameters","date":"2024-11-05","arxiv_id":"2411.03312","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/inference-optimal-vlms-need-only-one-visual#ran","syntology_url":"https://syntology.ai/paper/2411.03312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03312"}},"official":{"repos":["locuslab/llava-token-compression"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/logicity-advancing-neuro-symbolic-ai-with","slug":"logicity-advancing-neuro-symbolic-ai-with","title":"LogiCity: Advancing Neuro-Symbolic AI with Abstract Urban Simulation","date":"2024-11-01","arxiv_id":"2411.00773","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/logicity-advancing-neuro-symbolic-ai-with#ran","syntology_url":"https://syntology.ai/paper/2411.00773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00773"}},"official":{"repos":["Jaraxxus-Me/LogiCity"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bongard-in-wonderland-visual-puzzles-that","slug":"bongard-in-wonderland-visual-puzzles-that","title":"Bongard in Wonderland: Visual Puzzles that Still Make AI Go Mad?","date":"2024-10-25","arxiv_id":"2410.19546","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bongard-in-wonderland-visual-puzzles-that#ran","syntology_url":"https://syntology.ai/paper/2410.19546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19546"}},"official":{"repos":["ml-research/bongard-in-wonderland"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/distill-visual-chart-reasoning-ability-from","slug":"distill-visual-chart-reasoning-ability-from","title":"Distill Visual Chart Reasoning Ability from LLMs to MLLMs","date":"2024-10-24","arxiv_id":"2410.18798","repositories_listed":2,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/distill-visual-chart-reasoning-ability-from#ran","syntology_url":"https://syntology.ai/paper/2410.18798","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18798"}},"official":{"repos":["hewei2001/reachqa"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/humaneval-v-evaluating-visual-understanding","slug":"humaneval-v-evaluating-visual-understanding","title":"HumanEval-V: Evaluating Visual Understanding and Reasoning Abilities of Large Multimodal Models Through Coding Tasks","date":"2024-10-16","arxiv_id":"2410.12381","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/humaneval-v-evaluating-visual-understanding#ran","syntology_url":"https://syntology.ai/paper/2410.12381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12381"}},"official":{"repos":["HumanEval-V/HumanEval-V-Benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mctbench-multimodal-cognition-towards-text","slug":"mctbench-multimodal-cognition-towards-text","title":"MCTBench: Multimodal Cognition towards Text-Rich Visual Scenes Benchmark","date":"2024-10-15","arxiv_id":"2410.11538","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mctbench-multimodal-cognition-towards-text#ran","syntology_url":"https://syntology.ai/paper/2410.11538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11538"}},"official":{"repos":["xfey/mctbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tackling-the-abstraction-and-reasoning-corpus-1","slug":"tackling-the-abstraction-and-reasoning-corpus-1","title":"Tackling the Abstraction and Reasoning Corpus with Vision Transformers: the Importance of 2D Representation, Positions, and Objects","date":"2024-10-08","arxiv_id":"2410.06405","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tackling-the-abstraction-and-reasoning-corpus-1#ran","syntology_url":"https://syntology.ai/paper/2410.06405","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06405"}},"official":{"repos":["khalil-research/ViTARC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/finecops-ref-a-new-dataset-and-task-for-fine","slug":"finecops-ref-a-new-dataset-and-task-for-fine","title":"FineCops-Ref: A new Dataset and Task for Fine-Grained Compositional Referring Expression Comprehension","date":"2024-09-23","arxiv_id":"2409.14750","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finecops-ref-a-new-dataset-and-task-for-fine#ran","syntology_url":"https://syntology.ai/paper/2409.14750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.14750"}},"official":{"repos":["liujunzhuo/FineCops-Ref"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/journeybench-a-challenging-one-stop-vision","slug":"journeybench-a-challenging-one-stop-vision","title":"JourneyBench: A Challenging One-Stop Vision-Language Understanding Benchmark of Generated Images","date":"2024-09-19","arxiv_id":"2409.12953","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/journeybench-a-challenging-one-stop-vision#ran","syntology_url":"https://syntology.ai/paper/2409.12953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12953"}},"official":{"repos":["journeybench/journeybench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-determine-the-preferred-image","slug":"how-to-determine-the-preferred-image","title":"How to Determine the Preferred Image Distribution of a Black-Box Vision-Language Model?","date":"2024-09-03","arxiv_id":"2409.02253","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-to-determine-the-preferred-image#ran","syntology_url":"https://syntology.ai/paper/2409.02253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02253"}},"official":{"repos":["asgsaeid/cad_vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unibench-visual-reasoning-requires-rethinking","slug":"unibench-visual-reasoning-requires-rethinking","title":"UniBench: Visual Reasoning Requires Rethinking Vision-Language Beyond Scaling","date":"2024-08-09","arxiv_id":"2408.04810","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/unibench-visual-reasoning-requires-rethinking#ran","syntology_url":"https://syntology.ai/paper/2408.04810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04810"}},"official":{"repos":["facebookresearch/unibench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/kiva-kid-inspired-visual-analogies-for","slug":"kiva-kid-inspired-visual-analogies-for","title":"KiVA: Kid-inspired Visual Analogies for Testing Large Multimodal Models","date":"2024-07-25","arxiv_id":"2407.17773","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/kiva-kid-inspired-visual-analogies-for#ran","syntology_url":"https://syntology.ai/paper/2407.17773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17773"}},"official":{"repos":["ey242/kiva"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-self-instruct-synthetic-abstract","slug":"multimodal-self-instruct-synthetic-abstract","title":"Multimodal Self-Instruct: Synthetic Abstract Image and Visual Reasoning Instruction Using Language Model","date":"2024-07-09","arxiv_id":"2407.07053","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-self-instruct-synthetic-abstract#ran","syntology_url":"https://syntology.ai/paper/2407.07053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07053"}},"official":{"repos":["zwq2018/multi-modal-self-instruct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tokenpacker-efficient-visual-projector-for","slug":"tokenpacker-efficient-visual-projector-for","title":"TokenPacker: Efficient Visual Projector for Multimodal LLM","date":"2024-07-02","arxiv_id":"2407.02392","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tokenpacker-efficient-visual-projector-for#ran","syntology_url":"https://syntology.ai/paper/2407.02392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02392"}},"official":{"repos":["circleradon/tokenpacker"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/we-math-does-your-large-multimodal-model","slug":"we-math-does-your-large-multimodal-model","title":"We-Math: Does Your Large Multimodal Model Achieve Human-like Mathematical Reasoning?","date":"2024-07-01","arxiv_id":"2407.01284","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/we-math-does-your-large-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2407.01284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01284"}},"official":{"repos":["we-math/we-math"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/from-the-least-to-the-most-building-a-plug","slug":"from-the-least-to-the-most-building-a-plug","title":"From the Least to the Most: Building a Plug-and-Play Visual Reasoner via Data Synthesis","date":"2024-06-28","arxiv_id":"2406.19934","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":1,"n_instrument":7,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 7 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/from-the-least-to-the-most-building-a-plug#ran","syntology_url":"https://syntology.ai/paper/2406.19934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19934"}},"official":{"repos":["steven-ccq/visualreasoner"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/selective-vision-is-the-challenge-for-visual","slug":"selective-vision-is-the-challenge-for-visual","title":"Selective Vision is the Challenge for Visual Reasoning: A Benchmark for Visual Argument Understanding","date":"2024-06-27","arxiv_id":"2406.18925","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/selective-vision-is-the-challenge-for-visual#ran","syntology_url":"https://syntology.ai/paper/2406.18925","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18925"}},"official":{"repos":["jiwanchung/visargs"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/slot-state-space-models","slug":"slot-state-space-models","title":"Slot State Space Models","date":"2024-06-18","arxiv_id":"2406.12272","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/slot-state-space-models#ran","syntology_url":"https://syntology.ai/paper/2406.12272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12272"}},"official":{"repos":["jindongjiang/slotssms"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/clawmachine-fetching-visual-tokens-as-an","slug":"clawmachine-fetching-visual-tokens-as-an","title":"ClawMachine: Learning to Fetch Visual Tokens for Referential Comprehension","date":"2024-06-17","arxiv_id":"2406.11327","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/clawmachine-fetching-visual-tokens-as-an#ran","syntology_url":"https://syntology.ai/paper/2406.11327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11327"}},"official":{"repos":["martian422/clawmachine"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-concept-binder","slug":"neural-concept-binder","title":"Neural Concept Binder","date":"2024-06-14","arxiv_id":"2406.09949","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-concept-binder#ran","syntology_url":"https://syntology.ai/paper/2406.09949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09949"}},"official":{"repos":["ml-research/neuralconceptbinder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vdgd-mitigating-lvlm-hallucinations-in","slug":"vdgd-mitigating-lvlm-hallucinations-in","title":"Visual Description Grounding Reduces Hallucinations and Boosts Reasoning in LVLMs","date":"2024-05-24","arxiv_id":"2405.15683","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vdgd-mitigating-lvlm-hallucinations-in#ran","syntology_url":"https://syntology.ai/paper/2405.15683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15683"}},"official":{"repos":["sreyan88/vdgd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/mmcode-evaluating-multi-modal-code-large","slug":"mmcode-evaluating-multi-modal-code-large","title":"MMCode: Benchmarking Multimodal Large Language Models for Code Generation with Visually Rich Programming Problems","date":"2024-04-15","arxiv_id":"2404.09486","repositories_listed":3,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":16,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mmcode-evaluating-multi-modal-code-large#ran","syntology_url":"https://syntology.ai/paper/2404.09486","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09486"}},"official":{"repos":["happylkx/mmcode","likaixin2000/mmcode"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-embeddings-the-promise-of-visual-table","slug":"beyond-embeddings-the-promise-of-visual-table","title":"Beyond Embeddings: The Promise of Visual Table in Visual Reasoning","date":"2024-03-27","arxiv_id":"2403.18252","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":3,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 2 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-embeddings-the-promise-of-visual-table#ran","syntology_url":"https://syntology.ai/paper/2403.18252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18252"}},"official":{"repos":["lavi-lab/visual-table"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-prumerge-adaptive-token-reduction-for","slug":"llava-prumerge-adaptive-token-reduction-for","title":"LLaVA-PruMerge: Adaptive Token Reduction for Efficient Large Multimodal Models","date":"2024-03-22","arxiv_id":"2403.15388","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-prumerge-adaptive-token-reduction-for#ran","syntology_url":"https://syntology.ai/paper/2403.15388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15388"}},"official":null}},{"url":"/paper/hydra-a-hyper-agent-for-dynamic-compositional","slug":"hydra-a-hyper-agent-for-dynamic-compositional","title":"HYDRA: A Hyper Agent for Dynamic Compositional Visual Reasoning","date":"2024-03-19","arxiv_id":"2403.12884","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hydra-a-hyper-agent-for-dynamic-compositional#ran","syntology_url":"https://syntology.ai/paper/2403.12884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12884"}},"official":{"repos":["ControlNet/HYDRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/just-shift-it-test-time-prototype-shifting","slug":"just-shift-it-test-time-prototype-shifting","title":"Just Shift It: Test-Time Prototype Shifting for Zero-Shot Generalization with Vision-Language Models","date":"2024-03-19","arxiv_id":"2403.12952","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/just-shift-it-test-time-prototype-shifting#ran","syntology_url":"https://syntology.ai/paper/2403.12952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12952"}},"official":{"repos":["elaine-sui/tps"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/how-far-are-we-from-intelligent-visual","slug":"how-far-are-we-from-intelligent-visual","title":"How Far Are We from Intelligent Visual Deductive Reasoning?","date":"2024-03-07","arxiv_id":"2403.04732","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-far-are-we-from-intelligent-visual#ran","syntology_url":"https://syntology.ai/paper/2403.04732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04732"}},"official":{"repos":["apple/ml-rpm-bench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/slot-abstractors-toward-scalable-abstract","slug":"slot-abstractors-toward-scalable-abstract","title":"Slot Abstractors: Toward Scalable Abstract Visual Reasoning","date":"2024-03-06","arxiv_id":"2403.03458","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/slot-abstractors-toward-scalable-abstract#ran","syntology_url":"https://syntology.ai/paper/2403.03458","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03458"}},"official":{"repos":["shanka123/slot-abstractor","slotabstractor/slotabstractor"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/what-is-missing-in-multilingual-visual","slug":"what-is-missing-in-multilingual-visual","title":"What Is Missing in Multilingual Visual Reasoning and How to Fix It","date":"2024-03-03","arxiv_id":"2403.01404","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-is-missing-in-multilingual-visual#ran","syntology_url":"https://syntology.ai/paper/2403.01404","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01404"}},"official":{"repos":["yueqis/multilingual_visual_reasoning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-disentanglement-in-downstream","slug":"revisiting-disentanglement-in-downstream","title":"Revisiting Disentanglement in Downstream Tasks: A Study on Its Necessity for Abstract Visual Reasoning","date":"2024-03-01","arxiv_id":"2403.00352","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/revisiting-disentanglement-in-downstream#ran","syntology_url":"https://syntology.ai/paper/2403.00352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00352"}},"official":{"repos":["richard-coder-nai/disentanglement-lib-necessity"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stop-reasoning-when-multimodal-llms-with","slug":"stop-reasoning-when-multimodal-llms-with","title":"Stop Reasoning! When Multimodal LLM with Chain-of-Thought Reasoning Meets Adversarial Image","date":"2024-02-22","arxiv_id":"2402.14899","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stop-reasoning-when-multimodal-llms-with#ran","syntology_url":"https://syntology.ai/paper/2402.14899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14899"}},"official":{"repos":["aipenguin/stopreasoning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cogcom-train-large-vision-language-models","slug":"cogcom-train-large-vision-language-models","title":"CogCoM: Train Large Vision-Language Models Diving into Details through Chain of Manipulations","date":"2024-02-06","arxiv_id":"2402.04236","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cogcom-train-large-vision-language-models#ran","syntology_url":"https://syntology.ai/paper/2402.04236","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.04236"}},"official":{"repos":["thudm/cogcom"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contextual-evaluating-context-sensitive-text","slug":"contextual-evaluating-context-sensitive-text","title":"ConTextual: Evaluating Context-Sensitive Text-Rich Visual Reasoning in Large Multimodal Models","date":"2024-01-24","arxiv_id":"2401.13311","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contextual-evaluating-context-sensitive-text#ran","syntology_url":"https://syntology.ai/paper/2401.13311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13311"}},"official":{"repos":["rohan598/contextual"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompting-large-vision-language-models-for","slug":"prompting-large-vision-language-models-for","title":"Prompting Large Vision-Language Models for Compositional Reasoning","date":"2024-01-20","arxiv_id":"2401.11337","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/prompting-large-vision-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2401.11337","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11337"}},"official":{"repos":["tossowski/keycomp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/image-safeguarding-reasoning-with-conditional","slug":"image-safeguarding-reasoning-with-conditional","title":"Image Safeguarding: Reasoning with Conditional Vision Language Model and Obfuscating Unsafe Content Counterfactually","date":"2024-01-19","arxiv_id":"2401.11035","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/image-safeguarding-reasoning-with-conditional#ran","syntology_url":"https://syntology.ai/paper/2401.11035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11035"}},"official":{"repos":["secureaiautonomylab/conditionalvlm"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-generative-abstract-reasoning","slug":"towards-generative-abstract-reasoning","title":"Towards Generative Abstract Reasoning: Completing Raven's Progressive Matrix via Rule Abstraction and Selection","date":"2024-01-18","arxiv_id":"2401.09966","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-generative-abstract-reasoning#ran","syntology_url":"https://syntology.ai/paper/2401.09966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09966"}},"official":{"repos":["fudanvi/generative-abstract-reasoning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vcoder-versatile-vision-encoders-for","slug":"vcoder-versatile-vision-encoders-for","title":"VCoder: Versatile Vision Encoders for Multimodal Large Language Models","date":"2023-12-21","arxiv_id":"2312.14233","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vcoder-versatile-vision-encoders-for#ran","syntology_url":"https://syntology.ai/paper/2312.14233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14233"}},"official":{"repos":["shi-labs/vcoder"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/one-self-configurable-model-to-solve-many","slug":"one-self-configurable-model-to-solve-many","title":"One Self-Configurable Model to Solve Many Abstract Visual Reasoning Problems","date":"2023-12-15","arxiv_id":"2312.09997","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/one-self-configurable-model-to-solve-many#ran","syntology_url":"https://syntology.ai/paper/2312.09997","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09997"}},"official":{"repos":["mikomel/sal"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchlmm-benchmarking-cross-style-visual","slug":"benchlmm-benchmarking-cross-style-visual","title":"BenchLMM: Benchmarking Cross-style Visual Capability of Large Multimodal Models","date":"2023-12-05","arxiv_id":"2312.02896","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchlmm-benchmarking-cross-style-visual#ran","syntology_url":"https://syntology.ai/paper/2312.02896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02896"}},"official":{"repos":["aifeg/benchgpt","aifeg/benchlmm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/x-instructblip-a-framework-for-aligning-x","slug":"x-instructblip-a-framework-for-aligning-x","title":"X-InstructBLIP: A Framework for aligning X-Modal instruction-aware representations to LLMs and Emergent Cross-modal Reasoning","date":"2023-11-30","arxiv_id":"2311.18799","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/x-instructblip-a-framework-for-aligning-x#ran","syntology_url":"https://syntology.ai/paper/2311.18799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.18799"}},"official":{"repos":["artemisp/lavis-xinstructblip","salesforce/lavis"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/how-many-unicorns-are-in-this-image-a-safety","slug":"how-many-unicorns-are-in-this-image-a-safety","title":"How Many Unicorns Are in This Image? A Safety Evaluation Benchmark for Vision LLMs","date":"2023-11-27","arxiv_id":"2311.16101","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-many-unicorns-are-in-this-image-a-safety#ran","syntology_url":"https://syntology.ai/paper/2311.16101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16101"}},"official":{"repos":["ucsc-vlaa/vllm-safety-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmmu-a-massive-multi-discipline-multimodal","slug":"mmmu-a-massive-multi-discipline-multimodal","title":"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","date":"2023-11-27","arxiv_id":"2311.16502","repositories_listed":5,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mmmu-a-massive-multi-discipline-multimodal#ran","syntology_url":"https://syntology.ai/paper/2311.16502","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16502"}},"official":{"repos":["MMMU-Benchmark/MMMU"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/compositional-chain-of-thought-prompting-for","slug":"compositional-chain-of-thought-prompting-for","title":"Compositional Chain-of-Thought Prompting for Large Multimodal Models","date":"2023-11-27","arxiv_id":"2311.17076","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/compositional-chain-of-thought-prompting-for#ran","syntology_url":"https://syntology.ai/paper/2311.17076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17076"}},"official":{"repos":["chancharikmitra/ccot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/genome-generative-neuro-symbolic-visual","slug":"genome-generative-neuro-symbolic-visual","title":"GENOME: GenerativE Neuro-symbOlic visual reasoning by growing and reusing ModulEs","date":"2023-11-08","arxiv_id":"2311.04901","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/genome-generative-neuro-symbolic-visual#ran","syntology_url":"https://syntology.ai/paper/2311.04901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04901"}},"official":null}},{"url":"/paper/weakly-supervised-semantic-parsing-with-2","slug":"weakly-supervised-semantic-parsing-with-2","title":"Weakly Supervised Semantic Parsing with Execution-based Spurious Program Filtering","date":"2023-11-02","arxiv_id":"2311.01161","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/weakly-supervised-semantic-parsing-with-2#ran","syntology_url":"https://syntology.ai/paper/2311.01161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01161"}},"official":{"repos":["klee972/exec-filter"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/what-makes-for-good-visual-instructions","slug":"what-makes-for-good-visual-instructions","title":"What Makes for Good Visual Instructions? Synthesizing Complex Visual Reasoning Instructions for Visual Instruction Tuning","date":"2023-11-02","arxiv_id":"2311.01487","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/what-makes-for-good-visual-instructions#ran","syntology_url":"https://syntology.ai/paper/2311.01487","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01487"}},"official":{"repos":["rucaibox/comvint"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/myriad-large-multimodal-model-by-applying","slug":"myriad-large-multimodal-model-by-applying","title":"Myriad: Large Multimodal Model by Applying Vision Experts for Industrial Anomaly Detection","date":"2023-10-29","arxiv_id":"2310.19070","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":2,"n_no_contract":2,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 2 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/myriad-large-multimodal-model-by-applying#ran","syntology_url":"https://syntology.ai/paper/2310.19070","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19070"}},"official":{"repos":["tzjtatata/myriad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/what-s-left-concept-grounding-with-logic","slug":"what-s-left-concept-grounding-with-logic","title":"What's Left? Concept Grounding with Logic-Enhanced Foundation Models","date":"2023-10-24","arxiv_id":"2310.16035","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/what-s-left-concept-grounding-with-logic#ran","syntology_url":"https://syntology.ai/paper/2310.16035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16035"}},"official":{"repos":["joyhsu0504/left"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bongard-openworld-few-shot-reasoning-for-free","slug":"bongard-openworld-few-shot-reasoning-for-free","title":"Bongard-OpenWorld: Few-Shot Reasoning for Free-form Visual Concepts in the Real World","date":"2023-10-16","arxiv_id":"2310.10207","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bongard-openworld-few-shot-reasoning-for-free#ran","syntology_url":"https://syntology.ai/paper/2310.10207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10207"}},"official":{"repos":["joyjayng/Bongard-OpenWorld"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mmicl-empowering-vision-language-model-with","slug":"mmicl-empowering-vision-language-model-with","title":"MMICL: Empowering Vision-language Model with Multi-Modal In-Context Learning","date":"2023-09-14","arxiv_id":"2309.07915","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mmicl-empowering-vision-language-model-with#ran","syntology_url":"https://syntology.ai/paper/2309.07915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.07915"}},"official":{"repos":["haozhezhao/mic","pkunlp-icler/mic"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sparkles-unlocking-chats-across-multiple","slug":"sparkles-unlocking-chats-across-multiple","title":"Sparkles: Unlocking Chats Across Multiple Images for Multimodal Instruction-Following Models","date":"2023-08-31","arxiv_id":"2308.16463","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":3,"n_instrument":7,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 7 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/sparkles-unlocking-chats-across-multiple#ran","syntology_url":"https://syntology.ai/paper/2308.16463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.16463"}},"official":{"repos":["hypjudy/sparkles"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-pet-vision-and-language-parameter","slug":"vl-pet-vision-and-language-parameter","title":"VL-PET: Vision-and-Language Parameter-Efficient Tuning via Granularity Control","date":"2023-08-18","arxiv_id":"2308.09804","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vl-pet-vision-and-language-parameter#ran","syntology_url":"https://syntology.ai/paper/2308.09804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09804"}},"official":{"repos":["henryhzy/vl-pet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uni-nlx-unifying-textual-explanations-for","slug":"uni-nlx-unifying-textual-explanations-for","title":"Uni-NLX: Unifying Textual Explanations for Vision and Vision-Language Tasks","date":"2023-08-17","arxiv_id":"2308.09033","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uni-nlx-unifying-textual-explanations-for#ran","syntology_url":"https://syntology.ai/paper/2308.09033","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09033"}},"official":{"repos":["fawazsammani/uni-nlx","fawazsammani/nlxgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/3d-vista-pre-trained-transformer-for-3d","slug":"3d-vista-pre-trained-transformer-for-3d","title":"3D-VisTA: Pre-trained Transformer for 3D Vision and Text Alignment","date":"2023-08-08","arxiv_id":"2308.04352","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/3d-vista-pre-trained-transformer-for-3d#ran","syntology_url":"https://syntology.ai/paper/2308.04352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04352"}},"official":null}},{"url":"/paper/how-is-chatgpt-s-behavior-changing-over-time","slug":"how-is-chatgpt-s-behavior-changing-over-time","title":"How is ChatGPT's behavior changing over time?","date":"2023-07-18","arxiv_id":"2307.09009","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-is-chatgpt-s-behavior-changing-over-time#ran","syntology_url":"https://syntology.ai/paper/2307.09009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.09009"}},"official":{"repos":["lchen001/llmdrift"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learning-differentiable-logic-programs-for","slug":"learning-differentiable-logic-programs-for","title":"Learning Differentiable Logic Programs for Abstract Visual Reasoning","date":"2023-07-03","arxiv_id":"2307.00928","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/learning-differentiable-logic-programs-for#ran","syntology_url":"https://syntology.ai/paper/2307.00928","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.00928"}},"official":{"repos":["ml-research/neumann"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/stop-pre-training-adapt-visual-language","slug":"stop-pre-training-adapt-visual-language","title":"Stop Pre-Training: Adapt Visual-Language Models to Unseen Languages","date":"2023-06-29","arxiv_id":"2306.16774","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stop-pre-training-adapt-visual-language#ran","syntology_url":"https://syntology.ai/paper/2306.16774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16774"}},"official":{"repos":["yasminekaroui/clicotea"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/v-lol-a-diagnostic-dataset-for-visual-logical","slug":"v-lol-a-diagnostic-dataset-for-visual-logical","title":"V-LoL: A Diagnostic Dataset for Visual Logical Learning","date":"2023-06-13","arxiv_id":"2306.07743","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/v-lol-a-diagnostic-dataset-for-visual-logical#ran","syntology_url":"https://syntology.ai/paper/2306.07743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07743"}},"official":{"repos":["ml-research/vlol-dataset-gen"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/systematic-visual-reasoning-through-object-1","slug":"systematic-visual-reasoning-through-object-1","title":"Systematic Visual Reasoning through Object-Centric Relational Abstraction","date":"2023-06-04","arxiv_id":"2306.02500","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/systematic-visual-reasoning-through-object-1#ran","syntology_url":"https://syntology.ai/paper/2306.02500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02500"}},"official":{"repos":["shanka123/ocra"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/crossget-cross-guided-ensemble-of-tokens-for","slug":"crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","arxiv_id":"2305.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crossget-cross-guided-ensemble-of-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2305.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17455"}},"official":{"repos":["sdc17/crossget"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/measuring-progress-in-fine-grained-vision-and","slug":"measuring-progress-in-fine-grained-vision-and","title":"Measuring Progress in Fine-grained Vision-and-Language Understanding","date":"2023-05-12","arxiv_id":"2305.07558","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/measuring-progress-in-fine-grained-vision-and#ran","syntology_url":"https://syntology.ai/paper/2305.07558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.07558"}},"official":{"repos":["e-bug/fine-grained-evals"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/visual-instruction-tuning-1","slug":"visual-instruction-tuning-1","title":"Visual Instruction Tuning","date":"2023-04-17","arxiv_id":"2304.08485","repositories_listed":13,"syntology":{"n":51,"n_ran":16,"n_constructed":6,"n_ran_checked":8,"n_instrument":8,"n_unverified":35,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 8 where Syntology's instrument failed) · 35 unverified","sample_list":"/paper/visual-instruction-tuning-1#ran","syntology_url":"https://syntology.ai/paper/2304.08485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08485"}},"official":{"repos":["haotian-liu/LLaVA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["community","listed","named_in_paper","official"]}}},{"url":"/paper/your-diffusion-model-is-secretly-a-zero-shot","slug":"your-diffusion-model-is-secretly-a-zero-shot","title":"Your Diffusion Model is Secretly a Zero-Shot Classifier","date":"2023-03-28","arxiv_id":"2303.16203","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/your-diffusion-model-is-secretly-a-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2303.16203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16203"}},"official":{"repos":["diffusion-classifier/diffusion-classifier"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/equivariant-similarity-for-vision-language","slug":"equivariant-similarity-for-vision-language","title":"Equivariant Similarity for Vision-Language Foundation Models","date":"2023-03-25","arxiv_id":"2303.14465","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/equivariant-similarity-for-vision-language#ran","syntology_url":"https://syntology.ai/paper/2303.14465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14465"}},"official":{"repos":["wangt-cn/eqben"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/abstract-visual-reasoning-an-algebraic","slug":"abstract-visual-reasoning-an-algebraic","title":"Abstract Visual Reasoning: An Algebraic Approach for Solving Raven's Progressive Matrices","date":"2023-03-21","arxiv_id":"2303.11730","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/abstract-visual-reasoning-an-algebraic#ran","syntology_url":"https://syntology.ai/paper/2303.11730","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11730"}},"official":{"repos":["xu-jingyi/algebraicmr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chatgpt-asks-blip-2-answers-automatic","slug":"chatgpt-asks-blip-2-answers-automatic","title":"ChatGPT Asks, BLIP-2 Answers: Automatic Questioning Towards Enriched Visual Descriptions","date":"2023-03-12","arxiv_id":"2303.06594","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chatgpt-asks-blip-2-answers-automatic#ran","syntology_url":"https://syntology.ai/paper/2303.06594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06594"}},"official":{"repos":["vision-cair/chatcaptioner"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-reason-over-visual-objects","slug":"learning-to-reason-over-visual-objects","title":"Learning to reason over visual objects","date":"2023-03-03","arxiv_id":"2303.02260","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-reason-over-visual-objects#ran","syntology_url":"https://syntology.ai/paper/2303.02260","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.02260"}},"official":{"repos":["shanka123/stsn"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/blip-2-bootstrapping-language-image-pre","slug":"blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","arxiv_id":"2301.12597","repositories_listed":17,"syntology":{"n":8,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/blip-2-bootstrapping-language-image-pre#ran","syntology_url":"https://syntology.ai/paper/2301.12597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12597"}},"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/position-guided-text-prompt-for-vision","slug":"position-guided-text-prompt-for-vision","title":"Position-guided Text Prompt for Vision-Language Pre-training","date":"2022-12-19","arxiv_id":"2212.09737","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/position-guided-text-prompt-for-vision#ran","syntology_url":"https://syntology.ai/paper/2212.09737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09737"}},"official":{"repos":["sail-sg/ptp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"544d21dfc4946f24fd4624e2dd36d0c6484cbcf891d123ae96a9e2667d6babcc","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}