{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering/papers/ran/1","list_of":"/task/visual-question-answering","task":"Visual Question Answering (VQA)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":4,"rows_per_page":100,"rows":[1,100],"of":359,"counts":{"archive_papers_tagged":2167,"with_a_code_link":1039,"where_syntology_ran_a_sample":359,"not_listed_spam_title":0,"listed":2167,"listed_where_code_ran":359,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":287,"every_run_a_failure_of_syntologys_instrument":72,"listed_with_a_run_with_no_instrument_failure":287,"listed_every_run_a_failure_of_syntologys_instrument":72,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering/papers/ran/1","prev":null,"next":"/task/visual-question-answering/papers/ran/2","papers":[{"url":"/paper/slotpi-physics-informed-object-centric","slug":"slotpi-physics-informed-object-centric","title":"SlotPi: Physics-informed Object-centric Reasoning Models","date":"2025-06-12","arxiv_id":"2506.10778","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/slotpi-physics-informed-object-centric#ran","syntology_url":"https://syntology.ai/paper/2506.10778","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.10778"}},"official":{"repos":["intell-sci-comput/slotpi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diagnosing-and-mitigating-modality","slug":"diagnosing-and-mitigating-modality","title":"Diagnosing and Mitigating Modality Interference in Multimodal Large Language Models","date":"2025-05-26","arxiv_id":"2505.19616","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diagnosing-and-mitigating-modality#ran","syntology_url":"https://syntology.ai/paper/2505.19616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19616"}},"official":null}},{"url":"/paper/unifying-multimodal-large-language-model","slug":"unifying-multimodal-large-language-model","title":"Unifying Multimodal Large Language Model Capabilities and Modalities via Model Merging","date":"2025-05-26","arxiv_id":"2505.19892","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2505.19892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19892"}},"official":{"repos":["walkerworldpeace/mllmerging"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mineanybuild-benchmarking-spatial-planning","slug":"mineanybuild-benchmarking-spatial-planning","title":"MineAnyBuild: Benchmarking Spatial Planning for Open-world AI Agents","date":"2025-05-26","arxiv_id":"2505.20148","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mineanybuild-benchmarking-spatial-planning#ran","syntology_url":"https://syntology.ai/paper/2505.20148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20148"}},"official":{"repos":["mineanybuild/mineanybuild"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/are-vision-language-models-ready-for-clinical","slug":"are-vision-language-models-ready-for-clinical","title":"Are Vision Language Models Ready for Clinical Diagnosis? A 3D Medical Benchmark for Tumor-centric Visual Question Answering","date":"2025-05-25","arxiv_id":"2505.18915","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-vision-language-models-ready-for-clinical#ran","syntology_url":"https://syntology.ai/paper/2505.18915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18915"}},"official":{"repos":["schuture/deeptumorvqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/medagentboard-benchmarking-multi-agent","slug":"medagentboard-benchmarking-multi-agent","title":"MedAgentBoard: Benchmarking Multi-Agent Collaboration with Conventional Methods for Diverse Medical Tasks","date":"2025-05-18","arxiv_id":"2505.12371","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/medagentboard-benchmarking-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2505.12371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12371"}},"official":{"repos":["yhzhu99/medagentboard"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/omgm-orchestrate-multiple-granularities-and","slug":"omgm-orchestrate-multiple-granularities-and","title":"OMGM: Orchestrate Multiple Granularities and Modalities for Efficient Multimodal Retrieval","date":"2025-05-10","arxiv_id":"2505.07879","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/omgm-orchestrate-multiple-granularities-and#ran","syntology_url":"https://syntology.ai/paper/2505.07879","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07879"}},"official":null}},{"url":"/paper/adcare-vlm-leveraging-large-vision-language","slug":"adcare-vlm-leveraging-large-vision-language","title":"AdCare-VLM: Leveraging Large Vision Language Model (LVLM) to Monitor Long-Term Medication Adherence and Care","date":"2025-05-01","arxiv_id":"2505.00275","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/adcare-vlm-leveraging-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2505.00275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.00275"}},"official":{"repos":["asad14053/AdCare-VLM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/unlearning-sensitive-information-in","slug":"unlearning-sensitive-information-in","title":"Unlearning Sensitive Information in Multimodal LLMs: Benchmark and Attack-Defense Evaluation","date":"2025-05-01","arxiv_id":"2505.01456","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unlearning-sensitive-information-in#ran","syntology_url":"https://syntology.ai/paper/2505.01456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.01456"}},"official":{"repos":["vaidehi99/unlok-vqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-on-prompt-compression-for","slug":"an-empirical-study-on-prompt-compression-for","title":"An Empirical Study on Prompt Compression for Large Language Models","date":"2025-04-24","arxiv_id":"2505.00019","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-on-prompt-compression-for#ran","syntology_url":"https://syntology.ai/paper/2505.00019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.00019"}},"official":{"repos":["3DAgentWorld/Toolkit-for-Prompt-Compression"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/facebench-a-multi-view-multi-level-facial","slug":"facebench-a-multi-view-multi-level-facial","title":"FaceBench: A Multi-View Multi-Level Facial Attribute VQA Dataset for Benchmarking Face Perception MLLMs","date":"2025-03-27","arxiv_id":"2503.21457","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/facebench-a-multi-view-multi-level-facial#ran","syntology_url":"https://syntology.ai/paper/2503.21457","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21457"}},"official":{"repos":["cvi-szu/facebench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/med3dvlm-an-efficient-vision-language-model","slug":"med3dvlm-an-efficient-vision-language-model","title":"Med3DVLM: An Efficient Vision-Language Model for 3D Medical Image Analysis","date":"2025-03-25","arxiv_id":"2503.20047","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/med3dvlm-an-efficient-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2503.20047","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20047"}},"official":{"repos":["mirthai/med3dvlm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/marten-visual-question-answering-with-mask","slug":"marten-visual-question-answering-with-mask","title":"Marten: Visual Question Answering with Mask Generation for Multi-modal Document Understanding","date":"2025-03-18","arxiv_id":"2503.14140","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/marten-visual-question-answering-with-mask#ran","syntology_url":"https://syntology.ai/paper/2503.14140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.14140"}},"official":null}},{"url":"/paper/kvq-boosting-video-quality-assessment-via","slug":"kvq-boosting-video-quality-assessment-via","title":"KVQ: Boosting Video Quality Assessment via Saliency-guided Local Perception","date":"2025-03-13","arxiv_id":"2503.10259","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":5,"n_ran_checked":5,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"9 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/kvq-boosting-video-quality-assessment-via#ran","syntology_url":"https://syntology.ai/paper/2503.10259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10259"}},"official":{"repos":["qyp2000/kvq"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/segagent-exploring-pixel-understanding-1","slug":"segagent-exploring-pixel-understanding-1","title":"SegAgent: Exploring Pixel Understanding Capabilities in MLLMs by Imitating Human Annotator Trajectories","date":"2025-03-11","arxiv_id":"2503.08625","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/segagent-exploring-pixel-understanding-1#ran","syntology_url":"https://syntology.ai/paper/2503.08625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08625"}},"official":{"repos":["aim-uofa/SegAgent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-token-level-text-image-foundation-model-for","slug":"a-token-level-text-image-foundation-model-for","title":"A Token-level Text Image Foundation Model for Document Understanding","date":"2025-03-04","arxiv_id":"2503.02304","repositories_listed":0,"syntology":{"n":6,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-token-level-text-image-foundation-model-for#ran","syntology_url":"https://syntology.ai/paper/2503.02304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.02304"}},"official":null}},{"url":"/paper/qwen2-5-vl-technical-report","slug":"qwen2-5-vl-technical-report","title":"Qwen2.5-VL Technical Report","date":"2025-02-19","arxiv_id":"2502.13923","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/qwen2-5-vl-technical-report#ran","syntology_url":"https://syntology.ai/paper/2502.13923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13923"}},"official":null}},{"url":"/paper/re-align-aligning-vision-language-models-via","slug":"re-align-aligning-vision-language-models-via","title":"Re-Align: Aligning Vision Language Models via Retrieval-Augmented Direct Preference Optimization","date":"2025-02-18","arxiv_id":"2502.13146","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/re-align-aligning-vision-language-models-via#ran","syntology_url":"https://syntology.ai/paper/2502.13146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13146"}},"official":{"repos":["taco-group/re-align"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmunlearner-reformulating-multimodal-machine","slug":"mmunlearner-reformulating-multimodal-machine","title":"MMUnlearner: Reformulating Multimodal Machine Unlearning in the Era of Multimodal Large Language Models","date":"2025-02-16","arxiv_id":"2502.11051","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mmunlearner-reformulating-multimodal-machine#ran","syntology_url":"https://syntology.ai/paper/2502.11051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11051"}},"official":{"repos":["z1zs/mmunlearner"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-the-potential-of-encoder-free","slug":"exploring-the-potential-of-encoder-free","title":"Exploring the Potential of Encoder-free Architectures in 3D LMMs","date":"2025-02-13","arxiv_id":"2502.09620","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-the-potential-of-encoder-free#ran","syntology_url":"https://syntology.ai/paper/2502.09620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.09620"}},"official":{"repos":["ivan-tang-3d/enel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-mini-efficient-image-and-video-large","slug":"llava-mini-efficient-image-and-video-large","title":"LLaVA-Mini: Efficient Image and Video Large Multimodal Models with One Vision Token","date":"2025-01-07","arxiv_id":"2501.03895","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-mini-efficient-image-and-video-large#ran","syntology_url":"https://syntology.ai/paper/2501.03895","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.03895"}},"official":{"repos":["ictnlp/llava-mini"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalizing-from-simple-to-hard-visual","slug":"generalizing-from-simple-to-hard-visual","title":"Generalizing from SIMPLE to HARD Visual Reasoning: Can We Mitigate Modality Imbalance in VLMs?","date":"2025-01-05","arxiv_id":"2501.02669","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizing-from-simple-to-hard-visual#ran","syntology_url":"https://syntology.ai/paper/2501.02669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.02669"}},"official":{"repos":["princeton-pli/vlm_s2h"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/medcot-medical-chain-of-thought-via","slug":"medcot-medical-chain-of-thought-via","title":"MedCoT: Medical Chain of Thought via Hierarchical Expert","date":"2024-12-18","arxiv_id":"2412.13736","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/medcot-medical-chain-of-thought-via#ran","syntology_url":"https://syntology.ai/paper/2412.13736","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.13736"}},"official":{"repos":["jxliu-ai/medcot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-a-multimodal-large-language-model","slug":"towards-a-multimodal-large-language-model","title":"Towards a Multimodal Large Language Model with Pixel-Level Insight for Biomedicine","date":"2024-12-12","arxiv_id":"2412.09278","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-a-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2412.09278","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09278"}},"official":{"repos":["shawnhuang497/medplib"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lyra-an-efficient-and-speech-centric","slug":"lyra-an-efficient-and-speech-centric","title":"Lyra: An Efficient and Speech-Centric Framework for Omni-Cognition","date":"2024-12-12","arxiv_id":"2412.09501","repositories_listed":1,"syntology":{"n":19,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":4,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/lyra-an-efficient-and-speech-centric#ran","syntology_url":"https://syntology.ai/paper/2412.09501","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09501"}},"official":{"repos":["dvlab-research/Lyra"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-prompt-alignment-for-text-to-image","slug":"fast-prompt-alignment-for-text-to-image","title":"Fast Prompt Alignment for Text-to-Image Generation","date":"2024-12-11","arxiv_id":"2412.08639","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fast-prompt-alignment-for-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2412.08639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08639"}},"official":{"repos":["tiktok/fast_prompt_alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmedpo-aligning-medical-vision-language","slug":"mmedpo-aligning-medical-vision-language","title":"MMedPO: Aligning Medical Vision-Language Models with Clinical-Aware Multimodal Preference Optimization","date":"2024-12-09","arxiv_id":"2412.06141","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":8,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mmedpo-aligning-medical-vision-language#ran","syntology_url":"https://syntology.ai/paper/2412.06141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06141"}},"official":{"repos":["aiming-lab/mmedpo"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/expanding-performance-boundaries-of-open","slug":"expanding-performance-boundaries-of-open","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","date":"2024-12-06","arxiv_id":"2412.05271","repositories_listed":1,"syntology":{"n":9,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/expanding-performance-boundaries-of-open#ran","syntology_url":"https://syntology.ai/paper/2412.05271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05271"}},"official":{"repos":["opengvlab/internvl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/zoomeye-enhancing-multimodal-llms-with-human","slug":"zoomeye-enhancing-multimodal-llms-with-human","title":"ZoomEye: Enhancing Multimodal LLMs with Human-Like Zooming Capabilities through Tree-Based Image Exploration","date":"2024-11-25","arxiv_id":"2411.16044","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zoomeye-enhancing-multimodal-llms-with-human#ran","syntology_url":"https://syntology.ai/paper/2411.16044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16044"}},"official":{"repos":["om-ai-lab/ZoomEye"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/teaching-vlms-to-localize-specific-objects","slug":"teaching-vlms-to-localize-specific-objects","title":"Teaching VLMs to Localize Specific Objects from In-context Examples","date":"2024-11-20","arxiv_id":"2411.13317","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/teaching-vlms-to-localize-specific-objects#ran","syntology_url":"https://syntology.ai/paper/2411.13317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13317"}},"official":{"repos":["sivandoveh/iploc"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-multimodal-retrieval-augmented","slug":"benchmarking-multimodal-retrieval-augmented","title":"Benchmarking Multimodal Retrieval Augmented Generation with Dynamic VQA Dataset and Self-adaptive Planning Agent","date":"2024-11-05","arxiv_id":"2411.02937","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-multimodal-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2411.02937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02937"}},"official":{"repos":["alibaba-nlp/omnisearch"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-vision-language-model-unlearning","slug":"benchmarking-vision-language-model-unlearning","title":"Benchmarking Vision Language Model Unlearning via Fictitious Facial Identity Dataset","date":"2024-11-05","arxiv_id":"2411.03554","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-vision-language-model-unlearning#ran","syntology_url":"https://syntology.ai/paper/2411.03554","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03554"}},"official":{"repos":["safolab-wisc/fiubench"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/right-this-way-can-vlms-guide-us-to-see-more","slug":"right-this-way-can-vlms-guide-us-to-see-more","title":"Right this way: Can VLMs Guide Us to See More to Answer Questions?","date":"2024-11-01","arxiv_id":"2411.00394","repositories_listed":1,"syntology":{"n":17,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":13,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/right-this-way-can-vlms-guide-us-to-see-more#ran","syntology_url":"https://syntology.ai/paper/2411.00394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00394"}},"official":{"repos":["LeoLee7/Directional_guidance"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":13,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/autobench-v-can-large-vision-language-models","slug":"autobench-v-can-large-vision-language-models","title":"AutoBench-V: Can Large Vision-Language Models Benchmark Themselves?","date":"2024-10-28","arxiv_id":"2410.21259","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autobench-v-can-large-vision-language-models#ran","syntology_url":"https://syntology.ai/paper/2410.21259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21259"}},"official":{"repos":["wad3birch/AutoBench-V"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/infinity-mm-scaling-multimodal-performance","slug":"infinity-mm-scaling-multimodal-performance","title":"Infinity-MM: Scaling Multimodal Performance with Large-Scale and High-Quality Instruction Data","date":"2024-10-24","arxiv_id":"2410.18558","repositories_listed":4,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/infinity-mm-scaling-multimodal-performance#ran","syntology_url":"https://syntology.ai/paper/2410.18558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18558"}},"official":null}},{"url":"/paper/progressive-compositionality-in-text-to-image","slug":"progressive-compositionality-in-text-to-image","title":"Progressive Compositionality In Text-to-Image Generative Models","date":"2024-10-22","arxiv_id":"2410.16719","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/progressive-compositionality-in-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2410.16719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16719"}},"official":{"repos":["evansh666/evogen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/frontiers-in-intelligent-colonoscopy","slug":"frontiers-in-intelligent-colonoscopy","title":"Frontiers in Intelligent Colonoscopy","date":"2024-10-22","arxiv_id":"2410.17241","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/frontiers-in-intelligent-colonoscopy#ran","syntology_url":"https://syntology.ai/paper/2410.17241","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17241"}},"official":{"repos":["ai4colonoscopy/intelliscope"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/worldcuisines-a-massive-scale-benchmark-for","slug":"worldcuisines-a-massive-scale-benchmark-for","title":"WorldCuisines: A Massive-Scale Benchmark for Multilingual and Multicultural Visual Question Answering on Global Cuisines","date":"2024-10-16","arxiv_id":"2410.12705","repositories_listed":1,"syntology":{"n":17,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/worldcuisines-a-massive-scale-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2410.12705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12705"}},"official":{"repos":["worldcuisines/worldcuisines"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/mmed-rag-versatile-multimodal-rag-system-for","slug":"mmed-rag-versatile-multimodal-rag-system-for","title":"MMed-RAG: Versatile Multimodal RAG System for Medical Vision Language Models","date":"2024-10-16","arxiv_id":"2410.13085","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mmed-rag-versatile-multimodal-rag-system-for#ran","syntology_url":"https://syntology.ai/paper/2410.13085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13085"}},"official":{"repos":["richard-peng-xia/mmed-rag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/livexiv-a-multi-modal-live-benchmark-based-on","slug":"livexiv-a-multi-modal-live-benchmark-based-on","title":"LiveXiv -- A Multi-Modal Live Benchmark Based on Arxiv Papers Content","date":"2024-10-14","arxiv_id":"2410.10783","repositories_listed":1,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":16,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/livexiv-a-multi-modal-live-benchmark-based-on#ran","syntology_url":"https://syntology.ai/paper/2410.10783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10783"}},"official":{"repos":["nimrodshabtay/livexiv"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-foundation-models-for-3d-vision-how","slug":"towards-foundation-models-for-3d-vision-how","title":"Towards Foundation Models for 3D Vision: How Close Are We?","date":"2024-10-14","arxiv_id":"2410.10799","repositories_listed":2,"syntology":{"n":29,"n_ran":20,"n_constructed":0,"n_ran_checked":19,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":19,"n_pointer_only":1,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/towards-foundation-models-for-3d-vision-how#ran","syntology_url":"https://syntology.ai/paper/2410.10799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10799"}},"official":{"repos":["princeton-vl/uniqa-3d"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/dataenvgym-data-generation-agents-in-teacher","slug":"dataenvgym-data-generation-agents-in-teacher","title":"DataEnvGym: Data Generation Agents in Teacher Environments with Student Feedback","date":"2024-10-08","arxiv_id":"2410.06215","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dataenvgym-data-generation-agents-in-teacher#ran","syntology_url":"https://syntology.ai/paper/2410.06215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06215"}},"official":{"repos":["codezakh/dataenvgym"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mc-cot-a-modular-collaborative-cot-framework","slug":"mc-cot-a-modular-collaborative-cot-framework","title":"MC-CoT: A Modular Collaborative CoT Framework for Zero-shot Medical-VQA with LLM and MLLM Integration","date":"2024-10-06","arxiv_id":"2410.04521","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":15,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mc-cot-a-modular-collaborative-cot-framework#ran","syntology_url":"https://syntology.ai/paper/2410.04521","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04521"}},"official":{"repos":["thomaswei-cn/MC-CoT"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-hitchhikers-guide-to-fine-grained-face","slug":"a-hitchhikers-guide-to-fine-grained-face","title":"A Hitchhikers Guide to Fine-Grained Face Forgery Detection Using Common Sense Reasoning","date":"2024-10-01","arxiv_id":"2410.00485","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-hitchhikers-guide-to-fine-grained-face#ran","syntology_url":"https://syntology.ai/paper/2410.00485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00485"}},"official":{"repos":["NickyFot/HitchhikersGuide"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/t2vs-meet-vlms-a-scalable-multimodal-dataset","slug":"t2vs-meet-vlms-a-scalable-multimodal-dataset","title":"T2Vs Meet VLMs: A Scalable Multimodal Dataset for Visual Harmfulness Recognition","date":"2024-09-29","arxiv_id":"2409.19734","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/t2vs-meet-vlms-a-scalable-multimodal-dataset#ran","syntology_url":"https://syntology.ai/paper/2409.19734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.19734"}},"official":{"repos":["nctu-eva-lab/vhd11k"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mediconfusion-can-you-trust-your-ai","slug":"mediconfusion-can-you-trust-your-ai","title":"MediConfusion: Can you trust your AI radiologist? Probing the reliability of multimodal medical foundation models","date":"2024-09-23","arxiv_id":"2409.15477","repositories_listed":2,"syntology":{"n":28,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":16,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":28,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 16 unverified","sample_list":"/paper/mediconfusion-can-you-trust-your-ai#ran","syntology_url":"https://syntology.ai/paper/2409.15477","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.15477"}},"official":{"repos":["mshahabsepehri/mediconfusion","AIF4S/MediConfusion"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":16,"ran_from_kinds":["official"]}}},{"url":"/paper/journeybench-a-challenging-one-stop-vision","slug":"journeybench-a-challenging-one-stop-vision","title":"JourneyBench: A Challenging One-Stop Vision-Language Understanding Benchmark of Generated Images","date":"2024-09-19","arxiv_id":"2409.12953","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/journeybench-a-challenging-one-stop-vision#ran","syntology_url":"https://syntology.ai/paper/2409.12953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12953"}},"official":{"repos":["journeybench/journeybench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/qwen2-vl-enhancing-vision-language-model-s","slug":"qwen2-vl-enhancing-vision-language-model-s","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","date":"2024-09-18","arxiv_id":"2409.12191","repositories_listed":8,"syntology":{"n":12,"n_ran":12,"n_constructed":1,"n_ran_checked":10,"n_instrument":2,"n_unverified":0,"n_honours":4,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"12 ran (of which 1 constructed an object rather than computing a result; 10 with no instrument failure: 4 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/qwen2-vl-enhancing-vision-language-model-s#ran","syntology_url":"https://syntology.ai/paper/2409.12191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12191"}},"official":{"repos":["qwenlm/qwen2-vl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/less-is-more-a-simple-yet-effective-token","slug":"less-is-more-a-simple-yet-effective-token","title":"Less is More: A Simple yet Effective Token Reduction Method for Efficient Multi-modal LLMs","date":"2024-09-17","arxiv_id":"2409.10994","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/less-is-more-a-simple-yet-effective-token#ran","syntology_url":"https://syntology.ai/paper/2409.10994","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.10994"}},"official":{"repos":["freedomintelligence/trim"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/columbus-evaluating-cognitive-lateral","slug":"columbus-evaluating-cognitive-lateral","title":"COLUMBUS: Evaluating COgnitive Lateral Understanding through Multiple-choice reBUSes","date":"2024-09-06","arxiv_id":"2409.04053","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/columbus-evaluating-cognitive-lateral#ran","syntology_url":"https://syntology.ai/paper/2409.04053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.04053"}},"official":{"repos":["koen-47/columbus"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-determine-the-preferred-image","slug":"how-to-determine-the-preferred-image","title":"How to Determine the Preferred Image Distribution of a Black-Box Vision-Language Model?","date":"2024-09-03","arxiv_id":"2409.02253","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-to-determine-the-preferred-image#ran","syntology_url":"https://syntology.ai/paper/2409.02253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02253"}},"official":{"repos":["asgsaeid/cad_vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/med-pmc-medical-personalized-multi-modal","slug":"med-pmc-medical-personalized-multi-modal","title":"Med-PMC: Medical Personalized Multi-modal Consultation with a Proactive Ask-First-Observe-Next Paradigm","date":"2024-08-16","arxiv_id":"2408.08693","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/med-pmc-medical-personalized-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2408.08693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08693"}},"official":{"repos":["liuhc0428/med-pmc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-agents-as-fast-and-slow-thinkers","slug":"visual-agents-as-fast-and-slow-thinkers","title":"Visual Agents as Fast and Slow Thinkers","date":"2024-08-16","arxiv_id":"2408.08862","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-agents-as-fast-and-slow-thinkers#ran","syntology_url":"https://syntology.ai/paper/2408.08862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08862"}},"official":{"repos":["guangyans/sys2-llava"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/surgical-vqla-adversarial-contrastive","slug":"surgical-vqla-adversarial-contrastive","title":"Surgical-VQLA++: Adversarial Contrastive Learning for Calibrated Robust Visual Question-Localized Answering in Robotic Surgery","date":"2024-08-09","arxiv_id":"2408.04958","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/surgical-vqla-adversarial-contrastive#ran","syntology_url":"https://syntology.ai/paper/2408.04958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04958"}},"official":{"repos":["longbai1006/surgical-vqlaplus"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-02900","slug":"2408-02900","title":"MedTrinity-25M: A Large-scale Multimodal Dataset with Multigranular Annotations for Medicine","date":"2024-08-06","arxiv_id":"2408.02900","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-02900#ran","syntology_url":"https://syntology.ai/paper/2408.02900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.02900"}},"official":{"repos":["UCSC-VLAA/MedTrinity-25M"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-label-cluster-discrimination-for-visual","slug":"multi-label-cluster-discrimination-for-visual","title":"Multi-label Cluster Discrimination for Visual Representation Learning","date":"2024-07-24","arxiv_id":"2407.17331","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":7,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 7 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 7 samples that ran constructed an object rather than computing a result","sample_list":"/paper/multi-label-cluster-discrimination-for-visual#ran","syntology_url":"https://syntology.ai/paper/2407.17331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17331"}},"official":{"repos":["deepglint/unicom"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":7,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/frechet-video-motion-distance-a-metric-for","slug":"frechet-video-motion-distance-a-metric-for","title":"Fréchet Video Motion Distance: A Metric for Evaluating Motion Consistency in Videos","date":"2024-07-23","arxiv_id":"2407.16124","repositories_listed":2,"syntology":{"n":23,"n_ran":20,"n_constructed":0,"n_ran_checked":20,"n_instrument":0,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":18,"n_pointer_only":1,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 2 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/frechet-video-motion-distance-a-metric-for#ran","syntology_url":"https://syntology.ai/paper/2407.16124","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16124"}},"official":{"repos":["ljh0v0/fmd-frechet-motion-distance"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/visual-haystacks-answering-harder-questions","slug":"visual-haystacks-answering-harder-questions","title":"Visual Haystacks: A Vision-Centric Needle-In-A-Haystack Benchmark","date":"2024-07-18","arxiv_id":"2407.13766","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/visual-haystacks-answering-harder-questions#ran","syntology_url":"https://syntology.ai/paper/2407.13766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13766"}},"official":{"repos":["visual-haystacks/vhs_benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/spiqa-a-dataset-for-multimodal-question","slug":"spiqa-a-dataset-for-multimodal-question","title":"SPIQA: A Dataset for Multimodal Question Answering on Scientific Papers","date":"2024-07-12","arxiv_id":"2407.09413","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spiqa-a-dataset-for-multimodal-question#ran","syntology_url":"https://syntology.ai/paper/2407.09413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09413"}},"official":{"repos":["google/spiqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wsi-vqa-interpreting-whole-slide-images-by","slug":"wsi-vqa-interpreting-whole-slide-images-by","title":"WSI-VQA: Interpreting Whole Slide Images by Generative Visual Question Answering","date":"2024-07-08","arxiv_id":"2407.05603","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wsi-vqa-interpreting-whole-slide-images-by#ran","syntology_url":"https://syntology.ai/paper/2407.05603","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05603"}},"official":{"repos":["cpystan/wsi-vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rule-reliable-multimodal-rag-for-factuality","slug":"rule-reliable-multimodal-rag-for-factuality","title":"RULE: Reliable Multimodal RAG for Factuality in Medical Vision Language Models","date":"2024-07-06","arxiv_id":"2407.05131","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rule-reliable-multimodal-rag-for-factuality#ran","syntology_url":"https://syntology.ai/paper/2407.05131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05131"}},"official":{"repos":["richard-peng-xia/rule"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/minigpt-med-large-language-model-as-a-general","slug":"minigpt-med-large-language-model-as-a-general","title":"MiniGPT-Med: Large Language Model as a General Interface for Radiology Diagnosis","date":"2024-07-04","arxiv_id":"2407.04106","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minigpt-med-large-language-model-as-a-general#ran","syntology_url":"https://syntology.ai/paper/2407.04106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04106"}},"official":{"repos":["vision-cair/minigpt-med"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-bounding-box-is-worth-one-token","slug":"a-bounding-box-is-worth-one-token","title":"A Bounding Box is Worth One Token: Interleaving Layout and Text in a Large Language Model for Document Understanding","date":"2024-07-02","arxiv_id":"2407.01976","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-bounding-box-is-worth-one-token#ran","syntology_url":"https://syntology.ai/paper/2407.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01976"}},"official":{"repos":["laytextllm/laytextllm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tarsier-recipes-for-training-and-evaluating-1","slug":"tarsier-recipes-for-training-and-evaluating-1","title":"Tarsier: Recipes for Training and Evaluating Large Video Description Models","date":"2024-06-30","arxiv_id":"2407.00634","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tarsier-recipes-for-training-and-evaluating-1#ran","syntology_url":"https://syntology.ai/paper/2407.00634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00634"}},"official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/from-the-least-to-the-most-building-a-plug","slug":"from-the-least-to-the-most-building-a-plug","title":"From the Least to the Most: Building a Plug-and-Play Visual Reasoner via Data Synthesis","date":"2024-06-28","arxiv_id":"2406.19934","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":1,"n_instrument":7,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 7 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/from-the-least-to-the-most-building-a-plug#ran","syntology_url":"https://syntology.ai/paper/2406.19934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19934"}},"official":{"repos":["steven-ccq/visualreasoner"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/stllava-med-self-training-large-language-and","slug":"stllava-med-self-training-large-language-and","title":"STLLaVA-Med: Self-Training Large Language and Vision Assistant for Medical Question-Answering","date":"2024-06-28","arxiv_id":"2406.19973","repositories_listed":1,"syntology":{"n":19,"n_ran":9,"n_constructed":4,"n_ran_checked":5,"n_instrument":4,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"9 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/stllava-med-self-training-large-language-and#ran","syntology_url":"https://syntology.ai/paper/2406.19973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19973"}},"official":{"repos":["heliossun/stllava-med"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/huatuogpt-vision-towards-injecting-medical","slug":"huatuogpt-vision-towards-injecting-medical","title":"HuatuoGPT-Vision, Towards Injecting Medical Visual Knowledge into Multimodal LLMs at Scale","date":"2024-06-27","arxiv_id":"2406.19280","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/huatuogpt-vision-towards-injecting-medical#ran","syntology_url":"https://syntology.ai/paper/2406.19280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19280"}},"official":{"repos":["freedomintelligence/huatuogpt-vision"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/long-context-transfer-from-language-to-vision","slug":"long-context-transfer-from-language-to-vision","title":"Long Context Transfer from Language to Vision","date":"2024-06-24","arxiv_id":"2406.16852","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/long-context-transfer-from-language-to-vision#ran","syntology_url":"https://syntology.ai/paper/2406.16852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16852"}},"official":{"repos":["evolvinglmms-lab/longva"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/biomedical-visual-instruction-tuning-with","slug":"biomedical-visual-instruction-tuning-with","title":"Biomedical Visual Instruction Tuning with Clinician Preference Alignment","date":"2024-06-19","arxiv_id":"2406.13173","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/biomedical-visual-instruction-tuning-with#ran","syntology_url":"https://syntology.ai/paper/2406.13173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13173"}},"official":{"repos":["mao1207/BioMed-VITAL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learnable-in-context-vector-for-visual","slug":"learnable-in-context-vector-for-visual","title":"LIVE: Learnable In-Context Vector for Visual Question Answering","date":"2024-06-19","arxiv_id":"2406.13185","repositories_listed":2,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/learnable-in-context-vector-for-visual#ran","syntology_url":"https://syntology.ai/paper/2406.13185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13185"}},"official":{"repos":["forjadeforest/live-learnable-in-context-vector"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/autohallusion-automatic-generation-of","slug":"autohallusion-automatic-generation-of","title":"AutoHallusion: Automatic Generation of Hallucination Benchmarks for Vision-Language Models","date":"2024-06-16","arxiv_id":"2406.10900","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autohallusion-automatic-generation-of#ran","syntology_url":"https://syntology.ai/paper/2406.10900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10900"}},"official":{"repos":["wuxiyang1996/AutoHallusion"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/foodieqa-a-multimodal-dataset-for-fine","slug":"foodieqa-a-multimodal-dataset-for-fine","title":"FoodieQA: A Multimodal Dataset for Fine-Grained Understanding of Chinese Food Culture","date":"2024-06-16","arxiv_id":"2406.11030","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/foodieqa-a-multimodal-dataset-for-fine#ran","syntology_url":"https://syntology.ai/paper/2406.11030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11030"}},"official":{"repos":["lyan62/FoodieQA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/videollama-2-advancing-spatial-temporal","slug":"videollama-2-advancing-spatial-temporal","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","date":"2024-06-11","arxiv_id":"2406.07476","repositories_listed":3,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":7,"n_instrument":5,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":8,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/videollama-2-advancing-spatial-temporal#ran","syntology_url":"https://syntology.ai/paper/2406.07476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07476"}},"official":{"repos":["damo-nlp-sg/videollama2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/dragonfly-multi-resolution-zoom-supercharges","slug":"dragonfly-multi-resolution-zoom-supercharges","title":"Dragonfly: Multi-Resolution Zoom-In Encoding Enhances Vision-Language Models","date":"2024-06-03","arxiv_id":"2406.00977","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dragonfly-multi-resolution-zoom-supercharges#ran","syntology_url":"https://syntology.ai/paper/2406.00977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00977"}},"official":{"repos":["togethercomputer/dragonfly"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deco-decoupling-token-compression-from","slug":"deco-decoupling-token-compression-from","title":"DeCo: Decoupling Token Compression from Semantic Abstraction in Multimodal Large Language Models","date":"2024-05-31","arxiv_id":"2405.20985","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deco-decoupling-token-compression-from#ran","syntology_url":"https://syntology.ai/paper/2405.20985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20985"}},"official":{"repos":["yaolinli/deco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/instruction-guided-visual-masking","slug":"instruction-guided-visual-masking","title":"Instruction-Guided Visual Masking","date":"2024-05-30","arxiv_id":"2405.19783","repositories_listed":1,"syntology":{"n":28,"n_ran":20,"n_constructed":7,"n_ran_checked":11,"n_instrument":9,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"20 ran (of which 7 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 9 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/instruction-guided-visual-masking#ran","syntology_url":"https://syntology.ai/paper/2405.19783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19783"}},"official":{"repos":["2toinf/ivm"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":7,"n_ran_no_instrument_failure":11,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/lm4lv-a-frozen-large-language-model-for-low","slug":"lm4lv-a-frozen-large-language-model-for-low","title":"LM4LV: A Frozen Large Language Model for Low-level Vision Tasks","date":"2024-05-24","arxiv_id":"2405.15734","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lm4lv-a-frozen-large-language-model-for-low#ran","syntology_url":"https://syntology.ai/paper/2405.15734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15734"}},"official":{"repos":["bytetriper/lm4lv"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mtvqa-benchmarking-multilingual-text-centric","slug":"mtvqa-benchmarking-multilingual-text-centric","title":"MTVQA: Benchmarking Multilingual Text-Centric Visual Question Answering","date":"2024-05-20","arxiv_id":"2405.11985","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mtvqa-benchmarking-multilingual-text-centric#ran","syntology_url":"https://syntology.ai/paper/2405.11985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11985"}},"official":{"repos":["bytedance/MTVQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","slug":"cumo-scaling-multimodal-llm-with-co-upcycled","title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","date":"2024-05-09","arxiv_id":"2405.05949","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled#ran","syntology_url":"https://syntology.ai/paper/2405.05949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05949"}},"official":{"repos":["shi-labs/cumo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/omnidrive-a-holistic-llm-agent-framework-for","slug":"omnidrive-a-holistic-llm-agent-framework-for","title":"OmniDrive: A Holistic Vision-Language Dataset for Autonomous Driving with Counterfactual Reasoning","date":"2024-05-02","arxiv_id":"2405.01533","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/omnidrive-a-holistic-llm-agent-framework-for#ran","syntology_url":"https://syntology.ai/paper/2405.01533","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.01533"}},"official":{"repos":["nvlabs/omnidrive"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/boter-bootstrapping-knowledge-selection-and","slug":"boter-bootstrapping-knowledge-selection-and","title":"Self-Bootstrapped Visual-Language Model for Knowledge Selection and Question Answering","date":"2024-04-22","arxiv_id":"2404.13947","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/boter-bootstrapping-knowledge-selection-and#ran","syntology_url":"https://syntology.ai/paper/2404.13947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13947"}},"official":{"repos":["haodongze/self-ksel-qans"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-collaboration-strategy-for-llms-in","slug":"adaptive-collaboration-strategy-for-llms-in","title":"MDAgents: An Adaptive Collaboration of LLMs for Medical Decision-Making","date":"2024-04-22","arxiv_id":"2404.15155","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-collaboration-strategy-for-llms-in#ran","syntology_url":"https://syntology.ai/paper/2404.15155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15155"}},"official":{"repos":["mitmedialab/mdagents"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lapa-latent-prompt-assist-model-for-medical","slug":"lapa-latent-prompt-assist-model-for-medical","title":"LaPA: Latent Prompt Assist Model For Medical Visual Question Answering","date":"2024-04-19","arxiv_id":"2404.13039","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/lapa-latent-prompt-assist-model-for-medical#ran","syntology_url":"https://syntology.ai/paper/2404.13039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13039"}},"official":{"repos":["garygutc/lapa_model"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/moe-tinymed-mixture-of-experts-for-tiny","slug":"moe-tinymed-mixture-of-experts-for-tiny","title":"Med-MoE: Mixture of Domain-Specific Experts for Lightweight Medical Vision-Language Models","date":"2024-04-16","arxiv_id":"2404.10237","repositories_listed":2,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/moe-tinymed-mixture-of-experts-for-tiny#ran","syntology_url":"https://syntology.ai/paper/2404.10237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10237"}},"official":{"repos":["jiangsongtao/med-moe","jiangsongtao/tinymed"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/textcot-zoom-in-for-enhanced-multimodal-text","slug":"textcot-zoom-in-for-enhanced-multimodal-text","title":"TextCoT: Zoom In for Enhanced Multimodal Text-Rich Image Understanding","date":"2024-04-15","arxiv_id":"2404.09797","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textcot-zoom-in-for-enhanced-multimodal-text#ran","syntology_url":"https://syntology.ai/paper/2404.09797","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09797"}},"official":{"repos":["bzluan/textcot"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-visual-question-answering-through","slug":"enhancing-visual-question-answering-through","title":"Enhancing Visual Question Answering through Question-Driven Image Captions as Prompts","date":"2024-04-12","arxiv_id":"2404.08589","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-visual-question-answering-through#ran","syntology_url":"https://syntology.ai/paper/2404.08589","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08589"}},"official":{"repos":["ovguyo/captions-in-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ma-lmm-memory-augmented-large-multimodal","slug":"ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","arxiv_id":"2404.05726","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ma-lmm-memory-augmented-large-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.05726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05726"}},"official":{"repos":["boheumd/MA-LMM"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unsolvable-problem-detection-evaluating","slug":"unsolvable-problem-detection-evaluating","title":"Unsolvable Problem Detection: Evaluating Trustworthiness of Vision Language Models","date":"2024-03-29","arxiv_id":"2403.20331","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unsolvable-problem-detection-evaluating#ran","syntology_url":"https://syntology.ai/paper/2403.20331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20331"}},"official":{"repos":["atsumiyai/upd"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/quantifying-and-mitigating-unimodal-biases-in","slug":"quantifying-and-mitigating-unimodal-biases-in","title":"Quantifying and Mitigating Unimodal Biases in Multimodal Large Language Models: A Causal Perspective","date":"2024-03-27","arxiv_id":"2403.18346","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantifying-and-mitigating-unimodal-biases-in#ran","syntology_url":"https://syntology.ai/paper/2403.18346","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18346"}},"official":{"repos":["opencausalab/more"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-cot-unleashing-chain-of-thought","slug":"visual-cot-unleashing-chain-of-thought","title":"Visual CoT: Advancing Multi-Modal Language Models with a Comprehensive Dataset and Benchmark for Chain-of-Thought Reasoning","date":"2024-03-25","arxiv_id":"2403.16999","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":4,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 2 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-cot-unleashing-chain-of-thought#ran","syntology_url":"https://syntology.ai/paper/2403.16999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16999"}},"official":{"repos":["deepcs233/visual-cot"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/illusionvqa-a-challenging-optical-illusion","slug":"illusionvqa-a-challenging-optical-illusion","title":"IllusionVQA: A Challenging Optical Illusion Dataset for Vision Language Models","date":"2024-03-23","arxiv_id":"2403.15952","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/illusionvqa-a-challenging-optical-illusion#ran","syntology_url":"https://syntology.ai/paper/2403.15952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15952"}},"official":{"repos":["csebuetnlp/illusionvqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vid-tldr-training-free-token-merging-for","slug":"vid-tldr-training-free-token-merging-for","title":"vid-TLDR: Training Free Token merging for Light-weight Video Transformer","date":"2024-03-20","arxiv_id":"2403.13347","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vid-tldr-training-free-token-merging-for#ran","syntology_url":"https://syntology.ai/paper/2403.13347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13347"}},"official":{"repos":["mlvlab/vid-tldr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hydra-a-hyper-agent-for-dynamic-compositional","slug":"hydra-a-hyper-agent-for-dynamic-compositional","title":"HYDRA: A Hyper Agent for Dynamic Compositional Visual Reasoning","date":"2024-03-19","arxiv_id":"2403.12884","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hydra-a-hyper-agent-for-dynamic-compositional#ran","syntology_url":"https://syntology.ai/paper/2403.12884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12884"}},"official":{"repos":["ControlNet/HYDRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-icl-bench-the-devil-in-the-details-of","slug":"vl-icl-bench-the-devil-in-the-details-of","title":"VL-ICL Bench: The Devil in the Details of Multimodal In-Context Learning","date":"2024-03-19","arxiv_id":"2403.13164","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vl-icl-bench-the-devil-in-the-details-of#ran","syntology_url":"https://syntology.ai/paper/2403.13164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13164"}},"official":{"repos":["ys-zong/vl-icl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/textmonkey-an-ocr-free-large-multimodal-model","slug":"textmonkey-an-ocr-free-large-multimodal-model","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","date":"2024-03-07","arxiv_id":"2403.04473","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textmonkey-an-ocr-free-large-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2403.04473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04473"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-the-gap-between-2d-and-3d-visual","slug":"bridging-the-gap-between-2d-and-3d-visual","title":"Bridging the Gap between 2D and 3D Visual Question Answering: A Fusion Approach for 3D VQA","date":"2024-02-24","arxiv_id":"2402.15933","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-the-gap-between-2d-and-3d-visual#ran","syntology_url":"https://syntology.ai/paper/2402.15933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15933"}},"official":{"repos":["matthewdm0816/bridgeqa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-aware-evaluation-for-vision","slug":"uncertainty-aware-evaluation-for-vision","title":"Uncertainty-Aware Evaluation for Vision-Language Models","date":"2024-02-22","arxiv_id":"2402.14418","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/uncertainty-aware-evaluation-for-vision#ran","syntology_url":"https://syntology.ai/paper/2402.14418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14418"}},"official":{"repos":["ensec-ai/vlm-uncertainty-bench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/cognitive-visual-language-mapper-advancing","slug":"cognitive-visual-language-mapper-advancing","title":"Cognitive Visual-Language Mapper: Advancing Multimodal Comprehension with Enhanced Visual Knowledge Alignment","date":"2024-02-21","arxiv_id":"2402.13561","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cognitive-visual-language-mapper-advancing#ran","syntology_url":"https://syntology.ai/paper/2402.13561","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13561"}},"official":{"repos":["hitsz-tmg/cognitive-visual-language-mapper"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/collavo-crayon-large-language-and-vision","slug":"collavo-crayon-large-language-and-vision","title":"CoLLaVO: Crayon Large Language and Vision mOdel","date":"2024-02-17","arxiv_id":"2402.11248","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/collavo-crayon-large-language-and-vision#ran","syntology_url":"https://syntology.ai/paper/2402.11248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11248"}},"official":{"repos":["ByungKwanLee/CoLLaVO"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-modal-preference-alignment-remedies","slug":"multi-modal-preference-alignment-remedies","title":"Multi-modal Preference Alignment Remedies Degradation of Visual Instruction Tuning on Language Models","date":"2024-02-16","arxiv_id":"2402.10884","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-modal-preference-alignment-remedies#ran","syntology_url":"https://syntology.ai/paper/2402.10884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10884"}},"official":{"repos":["findalexli/mllm-dpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}}],"record_sha256":"8cf409f5e3ef51925059589d4c0f18bedb7b026517efdcf0103f811165f2761a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}