{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering-1/papers/4","list_of":"/task/visual-question-answering-1","task":"Visual Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":22,"rows_per_page":100,"rows":[301,400],"of":2177,"counts":{"archive_papers_tagged":2177,"with_a_code_link":1042,"where_syntology_ran_a_sample":378,"not_listed_spam_title":0,"listed":2177,"listed_where_code_ran":378,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":308,"every_run_a_failure_of_syntologys_instrument":70,"listed_with_a_run_with_no_instrument_failure":308,"listed_every_run_a_failure_of_syntologys_instrument":70,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering-1","prev":"/task/visual-question-answering-1/papers/3","next":"/task/visual-question-answering-1/papers/5","papers":[{"url":"/paper/unirs-unifying-multi-temporal-remote-sensing","slug":"unirs-unifying-multi-temporal-remote-sensing","title":"UniRS: Unifying Multi-temporal Remote Sensing Tasks through Vision Language Models","date":"2024-12-30","arxiv_id":"2412.20742","repositories_listed":1,"syntology":null},{"url":"/paper/hallucinogen-a-benchmark-for-evaluating","slug":"hallucinogen-a-benchmark-for-evaluating","title":"HALLUCINOGEN: A Benchmark for Evaluating Object Hallucination in Large Visual-Language Models","date":"2024-12-29","arxiv_id":"2412.20622","repositories_listed":1,"syntology":null},{"url":"/paper/linin-logic-integrated-neural-inference","slug":"linin-logic-integrated-neural-inference","title":"LININ: Logic Integrated Neural Inference Network for Explanatory Visual Question Answering","date":"2024-12-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-preference-data-synthetic","slug":"multimodal-preference-data-synthetic","title":"Multimodal Preference Data Synthetic Alignment with Reward Model","date":"2024-12-23","arxiv_id":"2412.17417","repositories_listed":1,"syntology":null},{"url":"/paper/silvar-speech-driven-multimodal-model-for","slug":"silvar-speech-driven-multimodal-model-for","title":"SilVar: Speech Driven Multimodal Model for Reasoning Visual Question Answering and Object Localization","date":"2024-12-21","arxiv_id":"2412.16771","repositories_listed":1,"syntology":null},{"url":"/paper/nesycoco-a-neuro-symbolic-concept-composer","slug":"nesycoco-a-neuro-symbolic-concept-composer","title":"NeSyCoCo: A Neuro-Symbolic Concept Composer for Compositional Generalization","date":"2024-12-20","arxiv_id":"2412.15588","repositories_listed":1,"syntology":null},{"url":"/paper/autotrust-benchmarking-trustworthiness-in","slug":"autotrust-benchmarking-trustworthiness-in","title":"AutoTrust: Benchmarking Trustworthiness in Large Vision Language Models for Autonomous Driving","date":"2024-12-19","arxiv_id":"2412.15206","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/autotrust-benchmarking-trustworthiness-in#ran","syntology_url":"https://syntology.ai/paper/2412.15206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15206"}},"official":{"repos":["taco-group/autotrust"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/defeasible-visual-entailment-benchmark","slug":"defeasible-visual-entailment-benchmark","title":"Defeasible Visual Entailment: Benchmark, Evaluator, and Reward-Driven Optimization","date":"2024-12-19","arxiv_id":"2412.16232","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-uncertainty-a-deep-dive-into","slug":"unveiling-uncertainty-a-deep-dive-into","title":"Unveiling Uncertainty: A Deep Dive into Calibration and Performance of Multimodal Large Language Models","date":"2024-12-19","arxiv_id":"2412.14660","repositories_listed":1,"syntology":null},{"url":"/paper/consistency-of-compositional-generalization","slug":"consistency-of-compositional-generalization","title":"Consistency of Compositional Generalization across Multiple Levels","date":"2024-12-18","arxiv_id":"2412.13636","repositories_listed":1,"syntology":null},{"url":"/paper/medcot-medical-chain-of-thought-via","slug":"medcot-medical-chain-of-thought-via","title":"MedCoT: Medical Chain of Thought via Hierarchical Expert","date":"2024-12-18","arxiv_id":"2412.13736","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/medcot-medical-chain-of-thought-via#ran","syntology_url":"https://syntology.ai/paper/2412.13736","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.13736"}},"official":{"repos":["jxliu-ai/medcot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/medmax-mixed-modal-instruction-tuning-for","slug":"medmax-mixed-modal-instruction-tuning-for","title":"MedMax: Mixed-Modal Instruction Tuning for Training Biomedical Assistants","date":"2024-12-17","arxiv_id":"2412.12661","repositories_listed":1,"syntology":null},{"url":"/paper/track-the-answer-extending-textvqa-from-image","slug":"track-the-answer-extending-textvqa-from-image","title":"Track the Answer: Extending TextVQA from Image to Video with Spatio-Temporal Clues","date":"2024-12-17","arxiv_id":"2412.12502","repositories_listed":1,"syntology":null},{"url":"/paper/visual-instruction-tuning-with-500x-fewer","slug":"visual-instruction-tuning-with-500x-fewer","title":"LLaVA Steering: Visual Instruction Tuning with 500x Fewer Parameters through Modality Linear Representation-Steering","date":"2024-12-16","arxiv_id":"2412.12359","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/visual-instruction-tuning-with-500x-fewer#ran","syntology_url":"https://syntology.ai/paper/2412.12359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12359"}},"official":{"repos":["bibisbar/LLaVA-Steering"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/deepseek-vl2-mixture-of-experts-vision","slug":"deepseek-vl2-mixture-of-experts-vision","title":"DeepSeek-VL2: Mixture-of-Experts Vision-Language Models for Advanced Multimodal Understanding","date":"2024-12-13","arxiv_id":"2412.10302","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deepseek-vl2-mixture-of-experts-vision#ran","syntology_url":"https://syntology.ai/paper/2412.10302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.10302"}},"official":{"repos":["deepseek-ai/deepseek-vl2"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/doe-1-closed-loop-autonomous-driving-with","slug":"doe-1-closed-loop-autonomous-driving-with","title":"Doe-1: Closed-Loop Autonomous Driving with Large World Model","date":"2024-12-12","arxiv_id":"2412.09627","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/doe-1-closed-loop-autonomous-driving-with#ran","syntology_url":"https://syntology.ai/paper/2412.09627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09627"}},"official":{"repos":["wzzheng/doe"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lyra-an-efficient-and-speech-centric","slug":"lyra-an-efficient-and-speech-centric","title":"Lyra: An Efficient and Speech-Centric Framework for Omni-Cognition","date":"2024-12-12","arxiv_id":"2412.09501","repositories_listed":1,"syntology":{"n":19,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":4,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/lyra-an-efficient-and-speech-centric#ran","syntology_url":"https://syntology.ai/paper/2412.09501","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09501"}},"official":{"repos":["dvlab-research/Lyra"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-a-multimodal-large-language-model","slug":"towards-a-multimodal-large-language-model","title":"Towards a Multimodal Large Language Model with Pixel-Level Insight for Biomedicine","date":"2024-12-12","arxiv_id":"2412.09278","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-a-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2412.09278","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09278"}},"official":{"repos":["shawnhuang497/medplib"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discrete-subgraph-sampling-for-interpretable","slug":"discrete-subgraph-sampling-for-interpretable","title":"Discrete Subgraph Sampling for Interpretable Graph based Visual Question Answering","date":"2024-12-11","arxiv_id":"2412.08263","repositories_listed":1,"syntology":null},{"url":"/paper/illusory-vqa-benchmarking-and-enhancing","slug":"illusory-vqa-benchmarking-and-enhancing","title":"Illusory VQA: Benchmarking and Enhancing Multimodal Models on Visual Illusions","date":"2024-12-11","arxiv_id":"2412.08169","repositories_listed":1,"syntology":null},{"url":"/paper/bimedix2-bio-medical-expert-lmm-for-diverse","slug":"bimedix2-bio-medical-expert-lmm-for-diverse","title":"BiMediX2: Bio-Medical EXpert LMM for Diverse Medical Modalities","date":"2024-12-10","arxiv_id":"2412.07769","repositories_listed":1,"syntology":null},{"url":"/paper/impact-a-large-scale-integrated-multimodal","slug":"impact-a-large-scale-integrated-multimodal","title":"IMPACT: A Large-scale Integrated Multimodal Patent Analysis and Creation Dataset for Design Patents","date":"2024-12-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mm-poe-multiple-choice-reasoning-via-process","slug":"mm-poe-multiple-choice-reasoning-via-process","title":"MM-PoE: Multiple Choice Reasoning via. Process of Elimination using Multi-Modal Models","date":"2024-12-10","arxiv_id":"2412.07148","repositories_listed":1,"syntology":null},{"url":"/paper/fm2ds-few-shot-multimodal-multihop-data","slug":"fm2ds-few-shot-multimodal-multihop-data","title":"FM2DS: Few-Shot Multimodal Multihop Data Synthesis with Knowledge Distillation for Question Answering","date":"2024-12-09","arxiv_id":"2412.07030","repositories_listed":1,"syntology":null},{"url":"/paper/provision-programmatically-scaling-vision","slug":"provision-programmatically-scaling-vision","title":"ProVision: Programmatically Scaling Vision-centric Instruction Data for Multimodal Language Models","date":"2024-12-09","arxiv_id":"2412.07012","repositories_listed":1,"syntology":{"n":15,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/provision-programmatically-scaling-vision#ran","syntology_url":"https://syntology.ai/paper/2412.07012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07012"}},"official":{"repos":["jieyuz2/provision"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/rsunivlm-a-unified-vision-language-model-for","slug":"rsunivlm-a-unified-vision-language-model-for","title":"RSUniVLM: A Unified Vision Language Model for Remote Sensing via Granularity-oriented Mixture of Experts","date":"2024-12-07","arxiv_id":"2412.05679","repositories_listed":1,"syntology":null},{"url":"/paper/taco-learning-multi-modal-action-models-with","slug":"taco-learning-multi-modal-action-models-with","title":"TACO: Learning Multi-modal Action Models with Synthetic Chains-of-Thought-and-Action","date":"2024-12-07","arxiv_id":"2412.05479","repositories_listed":1,"syntology":null},{"url":"/paper/expanding-performance-boundaries-of-open","slug":"expanding-performance-boundaries-of-open","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","date":"2024-12-06","arxiv_id":"2412.05271","repositories_listed":1,"syntology":{"n":9,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/expanding-performance-boundaries-of-open#ran","syntology_url":"https://syntology.ai/paper/2412.05271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05271"}},"official":{"repos":["opengvlab/internvl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/linvt-empower-your-image-level-large-language","slug":"linvt-empower-your-image-level-large-language","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","date":"2024-12-06","arxiv_id":"2412.05185","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":6,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/linvt-empower-your-image-level-large-language#ran","syntology_url":"https://syntology.ai/paper/2412.05185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05185"}},"official":{"repos":["gls0425/linvt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/mammoth-vl-eliciting-multimodal-reasoning","slug":"mammoth-vl-eliciting-multimodal-reasoning","title":"MAmmoTH-VL: Eliciting Multimodal Reasoning with Instruction Tuning at Scale","date":"2024-12-06","arxiv_id":"2412.05237","repositories_listed":1,"syntology":null},{"url":"/paper/flashsloth-lightning-multimodal-large","slug":"flashsloth-lightning-multimodal-large","title":"FlashSloth: Lightning Multimodal Large Language Models via Embedded Visual Compression","date":"2024-12-05","arxiv_id":"2412.04317","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/flashsloth-lightning-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2412.04317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.04317"}},"official":{"repos":["codefanw/flashsloth"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visionzip-longer-is-better-but-not-necessary","slug":"visionzip-longer-is-better-but-not-necessary","title":"VisionZip: Longer is Better but Not Necessary in Vision Language Models","date":"2024-12-05","arxiv_id":"2412.04467","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/visionzip-longer-is-better-but-not-necessary#ran","syntology_url":"https://syntology.ai/paper/2412.04467","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.04467"}},"official":{"repos":["dvlab-research/visionzip"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-stitch-in-time-saves-nine-small-vlm-is-a","slug":"a-stitch-in-time-saves-nine-small-vlm-is-a","title":"A Stitch in Time Saves Nine: Small VLM is a Precise Guidance for Accelerating Large VLMs","date":"2024-12-04","arxiv_id":"2412.03324","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":2,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-stitch-in-time-saves-nine-small-vlm-is-a#ran","syntology_url":"https://syntology.ai/paper/2412.03324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03324"}},"official":{"repos":["NUS-HPC-AI-Lab/SGL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/inst-it-boosting-multimodal-instance","slug":"inst-it-boosting-multimodal-instance","title":"Inst-IT: Boosting Multimodal Instance Understanding via Explicit Visual Prompt Instruction Tuning","date":"2024-12-04","arxiv_id":"2412.03565","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inst-it-boosting-multimodal-instance#ran","syntology_url":"https://syntology.ai/paper/2412.03565","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03565"}},"official":{"repos":["inst-it/inst-it"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/copy-move-forgery-detection-and-question","slug":"copy-move-forgery-detection-and-question","title":"Copy-Move Forgery Detection and Question Answering for Remote Sensing Image","date":"2024-12-03","arxiv_id":"2412.02575","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-llava-efficient-multimodal-large","slug":"dynamic-llava-efficient-multimodal-large","title":"Dynamic-LLaVA: Efficient Multimodal Large Language Models via Dynamic Vision-language Context Sparsification","date":"2024-12-01","arxiv_id":"2412.00876","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":4,"n_ran_checked":5,"n_instrument":8,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":0,"phrase":"13 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 8 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/dynamic-llava-efficient-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2412.00876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.00876"}},"official":{"repos":["osilly/dynamic_llava"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/dlava-document-language-and-vision-assistant","slug":"dlava-document-language-and-vision-assistant","title":"DLaVA: Document Language and Vision Assistant for Answer Localization with Enhanced Interpretability and Trustworthiness","date":"2024-11-29","arxiv_id":"2412.00151","repositories_listed":1,"syntology":null},{"url":"/paper/sure-vqa-systematic-understanding-of","slug":"sure-vqa-systematic-understanding-of","title":"SURE-VQA: Systematic Understanding of Robustness Evaluation in Medical VQA Tasks","date":"2024-11-29","arxiv_id":"2411.19688","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-information-flow-in-multimodal","slug":"cross-modal-information-flow-in-multimodal","title":"Cross-modal Information Flow in Multimodal Large Language Models","date":"2024-11-27","arxiv_id":"2411.18620","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cross-modal-information-flow-in-multimodal#ran","syntology_url":"https://syntology.ai/paper/2411.18620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18620"}},"official":{"repos":["FightingFighting/cross-modal-information-flow-in-MLLM"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/grounding-iqa-multimodal-language-grounding","slug":"grounding-iqa-multimodal-language-grounding","title":"Grounding-IQA: Multimodal Language Grounding Model for Image Quality Assessment","date":"2024-11-26","arxiv_id":"2411.17237","repositories_listed":1,"syntology":null},{"url":"/paper/path-rag-knowledge-guided-key-region","slug":"path-rag-knowledge-guided-key-region","title":"Path-RAG: Knowledge-Guided Key Region Retrieval for Open-ended Pathology Visual Question Answering","date":"2024-11-26","arxiv_id":"2411.17073","repositories_listed":1,"syntology":null},{"url":"/paper/augmenting-multimodal-llms-with-self","slug":"augmenting-multimodal-llms-with-self","title":"Augmenting Multimodal LLMs with Self-Reflective Tokens for Knowledge-based Visual Question Answering","date":"2024-11-25","arxiv_id":"2411.16863","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/augmenting-multimodal-llms-with-self#ran","syntology_url":"https://syntology.ai/paper/2411.16863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16863"}},"official":{"repos":["aimagelab/reflectiva"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zoomeye-enhancing-multimodal-llms-with-human","slug":"zoomeye-enhancing-multimodal-llms-with-human","title":"ZoomEye: Enhancing Multimodal LLMs with Human-Like Zooming Capabilities through Tree-Based Image Exploration","date":"2024-11-25","arxiv_id":"2411.16044","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zoomeye-enhancing-multimodal-llms-with-human#ran","syntology_url":"https://syntology.ai/paper/2411.16044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16044"}},"official":{"repos":["om-ai-lab/ZoomEye"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gmai-vl-gmai-vl-5-5m-a-large-vision-language","slug":"gmai-vl-gmai-vl-5-5m-a-large-vision-language","title":"GMAI-VL & GMAI-VL-5.5M: A Large Vision-Language Model and A Comprehensive Multimodal Dataset Towards General Medical AI","date":"2024-11-21","arxiv_id":"2411.14522","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gmai-vl-gmai-vl-5-5m-a-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2411.14522","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14522"}},"official":{"repos":["uni-medical/gmai-vl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-contexts-clarify-ambiguous-expressions","slug":"visual-contexts-clarify-ambiguous-expressions","title":"Visual Contexts Clarify Ambiguous Expressions: A Benchmark Dataset","date":"2024-11-21","arxiv_id":"2411.14137","repositories_listed":1,"syntology":null},{"url":"/paper/teaching-vlms-to-localize-specific-objects","slug":"teaching-vlms-to-localize-specific-objects","title":"Teaching VLMs to Localize Specific Objects from In-context Examples","date":"2024-11-20","arxiv_id":"2411.13317","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/teaching-vlms-to-localize-specific-objects#ran","syntology_url":"https://syntology.ai/paper/2411.13317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13317"}},"official":{"repos":["sivandoveh/iploc"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-of-medical-vision-and-language","slug":"a-survey-of-medical-vision-and-language","title":"A Survey of Medical Vision-and-Language Applications and Their Techniques","date":"2024-11-19","arxiv_id":"2411.12195","repositories_listed":1,"syntology":null},{"url":"/paper/mc-llava-multi-concept-personalized-vision","slug":"mc-llava-multi-concept-personalized-vision","title":"MC-LLaVA: Multi-Concept Personalized Vision-Language Model","date":"2024-11-18","arxiv_id":"2411.11706","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mc-llava-multi-concept-personalized-vision#ran","syntology_url":"https://syntology.ai/paper/2411.11706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.11706"}},"official":{"repos":["arctanxarc/mc-llava"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/quantifying-preferences-of-vision-language","slug":"quantifying-preferences-of-vision-language","title":"Value-Spectrum: Quantifying Preferences of Vision-Language Models via Value Decomposition in Social Media Contexts","date":"2024-11-18","arxiv_id":"2411.11479","repositories_listed":1,"syntology":null},{"url":"/paper/backdoormbti-a-backdoor-learning-multimodal","slug":"backdoormbti-a-backdoor-learning-multimodal","title":"BackdoorMBTI: A Backdoor Learning Multimodal Benchmark Tool Kit for Backdoor Defense Evaluation","date":"2024-11-17","arxiv_id":"2411.11006","repositories_listed":1,"syntology":null},{"url":"/paper/janusflow-harmonizing-autoregression-and","slug":"janusflow-harmonizing-autoregression-and","title":"JanusFlow: Harmonizing Autoregression and Rectified Flow for Unified Multimodal Understanding and Generation","date":"2024-11-12","arxiv_id":"2411.07975","repositories_listed":1,"syntology":null},{"url":"/paper/sparrowvqe-visual-question-explanation-for","slug":"sparrowvqe-visual-question-explanation-for","title":"SparrowVQE: Visual Question Explanation for Course Content Understanding","date":"2024-11-12","arxiv_id":"2411.07516","repositories_listed":1,"syntology":null},{"url":"/paper/vqa-2-visual-question-answering-for-video","slug":"vqa-2-visual-question-answering-for-video","title":"VQA$^2$: Visual Question Answering for Video Quality Assessment","date":"2024-11-06","arxiv_id":"2411.03795","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-multimodal-retrieval-augmented","slug":"benchmarking-multimodal-retrieval-augmented","title":"Benchmarking Multimodal Retrieval Augmented Generation with Dynamic VQA Dataset and Self-adaptive Planning Agent","date":"2024-11-05","arxiv_id":"2411.02937","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-multimodal-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2411.02937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02937"}},"official":{"repos":["alibaba-nlp/omnisearch"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/right-this-way-can-vlms-guide-us-to-see-more","slug":"right-this-way-can-vlms-guide-us-to-see-more","title":"Right this way: Can VLMs Guide Us to See More to Answer Questions?","date":"2024-11-01","arxiv_id":"2411.00394","repositories_listed":1,"syntology":{"n":17,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":13,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/right-this-way-can-vlms-guide-us-to-see-more#ran","syntology_url":"https://syntology.ai/paper/2411.00394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00394"}},"official":{"repos":["LeoLee7/Directional_guidance"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":13,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/nearest-neighbor-normalization-improves","slug":"nearest-neighbor-normalization-improves","title":"Nearest Neighbor Normalization Improves Multimodal Retrieval","date":"2024-10-31","arxiv_id":"2410.24114","repositories_listed":1,"syntology":null},{"url":"/paper/show-me-what-and-where-has-changed-question","slug":"show-me-what-and-where-has-changed-question","title":"Show Me What and Where has Changed? Question Answering and Grounding for Remote Sensing Change Detection","date":"2024-10-31","arxiv_id":"2410.23828","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/show-me-what-and-where-has-changed-question#ran","syntology_url":"https://syntology.ai/paper/2410.23828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23828"}},"official":{"repos":["like413/vista"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/are-vlms-really-blind","slug":"are-vlms-really-blind","title":"Are VLMs Really Blind","date":"2024-10-29","arxiv_id":"2410.22029","repositories_listed":1,"syntology":null},{"url":"/paper/autobench-v-can-large-vision-language-models","slug":"autobench-v-can-large-vision-language-models","title":"AutoBench-V: Can Large Vision-Language Models Benchmark Themselves?","date":"2024-10-28","arxiv_id":"2410.21259","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autobench-v-can-large-vision-language-models#ran","syntology_url":"https://syntology.ai/paper/2410.21259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21259"}},"official":{"repos":["wad3birch/AutoBench-V"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/few-shot-multimodal-explanation-for-visual","slug":"few-shot-multimodal-explanation-for-visual","title":"Few-Shot Multimodal Explanation for Visual Question Answering","date":"2024-10-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/visual-text-matters-improving-text-kvqa-with","slug":"visual-text-matters-improving-text-kvqa-with","title":"Visual Text Matters: Improving Text-KVQA with Visual Text Entity Knowledge-aware Large Multimodal Assistant","date":"2024-10-24","arxiv_id":"2410.19144","repositories_listed":1,"syntology":null},{"url":"/paper/adem-vl-adaptive-and-embedded-fusion-for","slug":"adem-vl-adaptive-and-embedded-fusion-for","title":"ADEM-VL: Adaptive and Embedded Fusion for Efficient Vision-Language Tuning","date":"2024-10-23","arxiv_id":"2410.17779","repositories_listed":1,"syntology":null},{"url":"/paper/progressive-compositionality-in-text-to-image","slug":"progressive-compositionality-in-text-to-image","title":"Progressive Compositionality In Text-to-Image Generative Models","date":"2024-10-22","arxiv_id":"2410.16719","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/progressive-compositionality-in-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2410.16719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16719"}},"official":{"repos":["evansh666/evogen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/griffon-g-bridging-vision-language-and-vision","slug":"griffon-g-bridging-vision-language-and-vision","title":"Griffon-G: Bridging Vision-Language and Vision-Centric Tasks via Large Multimodal Models","date":"2024-10-21","arxiv_id":"2410.16163","repositories_listed":1,"syntology":null},{"url":"/paper/crope-evaluating-in-context-adaptation-of","slug":"crope-evaluating-in-context-adaptation-of","title":"CROPE: Evaluating In-Context Adaptation of Vision and Language Models to Culture-Specific Concepts","date":"2024-10-20","arxiv_id":"2410.15453","repositories_listed":1,"syntology":null},{"url":"/paper/multichartqa-benchmarking-vision-language","slug":"multichartqa-benchmarking-vision-language","title":"MultiChartQA: Benchmarking Vision-Language Models on Multi-Chart Problems","date":"2024-10-18","arxiv_id":"2410.14179","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multichartqa-benchmarking-vision-language#ran","syntology_url":"https://syntology.ai/paper/2410.14179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14179"}},"official":{"repos":["zivenzhu/multi-chart-qa"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/viconsformer-constituting-meaningful-phrases","slug":"viconsformer-constituting-meaningful-phrases","title":"ViConsFormer: Constituting Meaningful Phrases of Scene Texts using Transformer-based Method in Vietnamese Text-based Visual Question Answering","date":"2024-10-18","arxiv_id":"2410.14132","repositories_listed":1,"syntology":null},{"url":"/paper/help-me-identify-is-an-llm-vqa-system-all-we","slug":"help-me-identify-is-an-llm-vqa-system-all-we","title":"Help Me Identify: Is an LLM+VQA System All We Need to Identify Visual Concepts?","date":"2024-10-17","arxiv_id":"2410.13651","repositories_listed":1,"syntology":null},{"url":"/paper/janus-decoupling-visual-encoding-for-unified","slug":"janus-decoupling-visual-encoding-for-unified","title":"Janus: Decoupling Visual Encoding for Unified Multimodal Understanding and Generation","date":"2024-10-17","arxiv_id":"2410.13848","repositories_listed":1,"syntology":null},{"url":"/paper/vividmed-vision-language-model-with-versatile","slug":"vividmed-vision-language-model-with-versatile","title":"VividMed: Vision Language Model with Versatile Visual Grounding for Medicine","date":"2024-10-16","arxiv_id":"2410.12694","repositories_listed":1,"syntology":null},{"url":"/paper/worldcuisines-a-massive-scale-benchmark-for","slug":"worldcuisines-a-massive-scale-benchmark-for","title":"WorldCuisines: A Massive-Scale Benchmark for Multilingual and Multicultural Visual Question Answering on Global Cuisines","date":"2024-10-16","arxiv_id":"2410.12705","repositories_listed":1,"syntology":{"n":17,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/worldcuisines-a-massive-scale-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2410.12705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12705"}},"official":{"repos":["worldcuisines/worldcuisines"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/difficult-task-yes-but-simple-task-no","slug":"difficult-task-yes-but-simple-task-no","title":"Difficult Task Yes but Simple Task No: Unveiling the Laziness in Multimodal LLMs","date":"2024-10-15","arxiv_id":"2410.11437","repositories_listed":1,"syntology":null},{"url":"/paper/mmfuser-multimodal-multi-layer-feature-fuser","slug":"mmfuser-multimodal-multi-layer-feature-fuser","title":"MMFuser: Multimodal Multi-Layer Feature Fuser for Fine-Grained Vision-Language Understanding","date":"2024-10-15","arxiv_id":"2410.11829","repositories_listed":1,"syntology":null},{"url":"/paper/declarative-knowledge-distillation-from-large","slug":"declarative-knowledge-distillation-from-large","title":"Declarative Knowledge Distillation from Large Language Models for Visual Question Answering Datasets","date":"2024-10-12","arxiv_id":"2410.09428","repositories_listed":1,"syntology":null},{"url":"/paper/skipping-computations-in-multimodal-llms","slug":"skipping-computations-in-multimodal-llms","title":"Skipping Computations in Multimodal LLMs","date":"2024-10-12","arxiv_id":"2410.09454","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-commonsense-reasoning-over-machine","slug":"zero-shot-commonsense-reasoning-over-machine","title":"Zero-shot Commonsense Reasoning over Machine Imagination","date":"2024-10-12","arxiv_id":"2410.09329","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-multimodal-evaluation-with-flexible","slug":"dynamic-multimodal-evaluation-with-flexible","title":"Dynamic Multimodal Evaluation with Flexible Complexity by Vision-Language Bootstrapping","date":"2024-10-11","arxiv_id":"2410.08695","repositories_listed":1,"syntology":null},{"url":"/paper/voxelprompt-a-vision-language-agent-for","slug":"voxelprompt-a-vision-language-agent-for","title":"VoxelPrompt: A Vision-Language Agent for Grounded Medical Image Analysis","date":"2024-10-10","arxiv_id":"2410.08397","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/voxelprompt-a-vision-language-agent-for#ran","syntology_url":"https://syntology.ai/paper/2410.08397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08397"}},"official":{"repos":["dalcalab/voxel"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deciphering-cross-modal-alignment-in-large","slug":"deciphering-cross-modal-alignment-in-large","title":"Deciphering Cross-Modal Alignment in Large Vision-Language Models with Modality Integration Rate","date":"2024-10-09","arxiv_id":"2410.07167","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deciphering-cross-modal-alignment-in-large#ran","syntology_url":"https://syntology.ai/paper/2410.07167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07167"}},"official":{"repos":["shikiw/modality-integration-rate"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/core-tokensets-for-data-efficient-sequential","slug":"core-tokensets-for-data-efficient-sequential","title":"Core Tokensets for Data-efficient Sequential Training of Transformers","date":"2024-10-08","arxiv_id":"2410.05800","repositories_listed":1,"syntology":null},{"url":"/paper/ervqa-a-dataset-to-benchmark-the-readiness-of","slug":"ervqa-a-dataset-to-benchmark-the-readiness-of","title":"ERVQA: A Dataset to Benchmark the Readiness of Large Vision Language Models in Hospital Environments","date":"2024-10-08","arxiv_id":"2410.06420","repositories_listed":1,"syntology":null},{"url":"/paper/llaca-multimodal-large-language-continual","slug":"llaca-multimodal-large-language-continual","title":"Large Continual Instruction Assistant","date":"2024-10-08","arxiv_id":"2410.10868","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llaca-multimodal-large-language-continual#ran","syntology_url":"https://syntology.ai/paper/2410.10868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10868"}},"official":{"repos":["jingyangqiao/coin"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-large-language-models-and-tunings","slug":"multimodal-large-language-models-and-tunings","title":"Multimodal Large Language Models and Tunings: Vision, Language, Sensors, Audio, and Beyond","date":"2024-10-08","arxiv_id":"2410.05608","repositories_listed":1,"syntology":null},{"url":"/paper/teochat-a-large-vision-language-assistant-for","slug":"teochat-a-large-vision-language-assistant-for","title":"TEOChat: A Large Vision-Language Assistant for Temporal Earth Observation Data","date":"2024-10-08","arxiv_id":"2410.06234","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":5,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/teochat-a-large-vision-language-assistant-for#ran","syntology_url":"https://syntology.ai/paper/2410.06234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06234"}},"official":{"repos":["ermongroup/teochat"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/actiview-evaluating-active-perception-ability","slug":"actiview-evaluating-active-perception-ability","title":"ActiView: Evaluating Active Perception Ability for Multimodal Large Language Models","date":"2024-10-07","arxiv_id":"2410.04659","repositories_listed":1,"syntology":null},{"url":"/paper/mc-cot-a-modular-collaborative-cot-framework","slug":"mc-cot-a-modular-collaborative-cot-framework","title":"MC-CoT: A Modular Collaborative CoT Framework for Zero-shot Medical-VQA with LLM and MLLM Integration","date":"2024-10-06","arxiv_id":"2410.04521","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":15,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mc-cot-a-modular-collaborative-cot-framework#ran","syntology_url":"https://syntology.ai/paper/2410.04521","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04521"}},"official":{"repos":["thomaswei-cn/MC-CoT"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tubench-benchmarking-large-vision-language","slug":"tubench-benchmarking-large-vision-language","title":"TUBench: Benchmarking Large Vision-Language Models on Trustworthiness with Unanswerable Questions","date":"2024-10-05","arxiv_id":"2410.04107","repositories_listed":1,"syntology":null},{"url":"/paper/a-hitchhikers-guide-to-fine-grained-face","slug":"a-hitchhikers-guide-to-fine-grained-face","title":"A Hitchhikers Guide to Fine-Grained Face Forgery Detection Using Common Sense Reasoning","date":"2024-10-01","arxiv_id":"2410.00485","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-hitchhikers-guide-to-fine-grained-face#ran","syntology_url":"https://syntology.ai/paper/2410.00485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00485"}},"official":{"repos":["NickyFot/HitchhikersGuide"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/babelbench-an-omni-benchmark-for-code-driven","slug":"babelbench-an-omni-benchmark-for-code-driven","title":"BabelBench: An Omni Benchmark for Code-Driven Analysis of Multimodal and Multistructured Data","date":"2024-10-01","arxiv_id":"2410.00773","repositories_listed":1,"syntology":null},{"url":"/paper/unleashing-the-potentials-of-likelihood","slug":"unleashing-the-potentials-of-likelihood","title":"Unleashing the Potentials of Likelihood Composition for Multi-modal Language Models","date":"2024-10-01","arxiv_id":"2410.00363","repositories_listed":1,"syntology":null},{"url":"/paper/world-to-code-multi-modal-data-generation-via","slug":"world-to-code-multi-modal-data-generation-via","title":"World to Code: Multi-modal Data Generation via Self-Instructed Compositional Captioning and Filtering","date":"2024-09-30","arxiv_id":"2409.20424","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/world-to-code-multi-modal-data-generation-via#ran","syntology_url":"https://syntology.ai/paper/2409.20424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20424"}},"official":{"repos":["foundation-multimodal-models/world2code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/t2vs-meet-vlms-a-scalable-multimodal-dataset","slug":"t2vs-meet-vlms-a-scalable-multimodal-dataset","title":"T2Vs Meet VLMs: A Scalable Multimodal Dataset for Visual Harmfulness Recognition","date":"2024-09-29","arxiv_id":"2409.19734","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/t2vs-meet-vlms-a-scalable-multimodal-dataset#ran","syntology_url":"https://syntology.ai/paper/2409.19734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.19734"}},"official":{"repos":["nctu-eva-lab/vhd11k"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/uni-med-a-unified-medical-generalist","slug":"uni-med-a-unified-medical-generalist","title":"Uni-Med: A Unified Medical Generalist Foundation Model For Multi-Task Learning Via Connector-MoE","date":"2024-09-26","arxiv_id":"2409.17508","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-hallucination-mitigation-framework","slug":"a-unified-hallucination-mitigation-framework","title":"A Unified Hallucination Mitigation Framework for Large Vision-Language Models","date":"2024-09-24","arxiv_id":"2409.16494","repositories_listed":1,"syntology":null},{"url":"/paper/phantom-of-latent-for-large-language-and","slug":"phantom-of-latent-for-large-language-and","title":"Phantom of Latent for Large Language and Vision Models","date":"2024-09-23","arxiv_id":"2409.14713","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/phantom-of-latent-for-large-language-and#ran","syntology_url":"https://syntology.ai/paper/2409.14713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.14713"}},"official":{"repos":["byungkwanlee/phantom"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-image-hallucination-in-text-to","slug":"evaluating-image-hallucination-in-text-to","title":"Evaluating Image Hallucination in Text-to-Image Generation with Question-Answering","date":"2024-09-19","arxiv_id":"2409.12784","repositories_listed":1,"syntology":null},{"url":"/paper/cast-cross-modal-alignment-similarity-test","slug":"cast-cross-modal-alignment-similarity-test","title":"CAST: Cross-modal Alignment Similarity Test for Vision Language Models","date":"2024-09-17","arxiv_id":"2409.11007","repositories_listed":1,"syntology":null},{"url":"/paper/less-is-more-a-simple-yet-effective-token","slug":"less-is-more-a-simple-yet-effective-token","title":"Less is More: A Simple yet Effective Token Reduction Method for Efficient Multi-modal LLMs","date":"2024-09-17","arxiv_id":"2409.10994","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/less-is-more-a-simple-yet-effective-token#ran","syntology_url":"https://syntology.ai/paper/2409.10994","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.10994"}},"official":{"repos":["freedomintelligence/trim"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/guiding-vision-language-model-selection-for","slug":"guiding-vision-language-model-selection-for","title":"Guiding Vision-Language Model Selection for Visual Question-Answering Across Tasks, Domains, and Knowledge Types","date":"2024-09-14","arxiv_id":"2409.09269","repositories_listed":1,"syntology":null},{"url":"/paper/one-missing-piece-in-vision-and-language-a","slug":"one-missing-piece-in-vision-and-language-a","title":"One missing piece in Vision and Language: A Survey on Comics Understanding","date":"2024-09-14","arxiv_id":"2409.09502","repositories_listed":1,"syntology":null}],"record_sha256":"f9c1a9bd067185e873cf4c84df88d478a9f7409396bbc15dfdd5b50be064ef57","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}