{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering-1/papers/5","list_of":"/task/visual-question-answering-1","task":"Visual Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":22,"rows_per_page":100,"rows":[401,500],"of":2177,"counts":{"archive_papers_tagged":2177,"with_a_code_link":1042,"where_syntology_ran_a_sample":378,"not_listed_spam_title":0,"listed":2177,"listed_where_code_ran":378,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":308,"every_run_a_failure_of_syntologys_instrument":70,"listed_with_a_run_with_no_instrument_failure":308,"listed_every_run_a_failure_of_syntologys_instrument":70,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering-1","prev":"/task/visual-question-answering-1/papers/4","next":"/task/visual-question-answering-1/papers/6","papers":[{"url":"/paper/eyeclip-a-visual-language-foundation-model","slug":"eyeclip-a-visual-language-foundation-model","title":"EyeCLIP: A visual-language foundation model for multi-modal ophthalmic image analysis","date":"2024-09-10","arxiv_id":"2409.06644","repositories_listed":1,"syntology":null},{"url":"/paper/alt-moe-multimodal-alignment-via-alternating","slug":"alt-moe-multimodal-alignment-via-alternating","title":"M3-Jepa: Multimodal Alignment via Multi-directional MoE based on the JEPA framework","date":"2024-09-09","arxiv_id":"2409.05929","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alt-moe-multimodal-alignment-via-alternating#ran","syntology_url":"https://syntology.ai/paper/2409.05929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.05929"}},"official":{"repos":["HongyangLL/M3-JEPA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/columbus-evaluating-cognitive-lateral","slug":"columbus-evaluating-cognitive-lateral","title":"COLUMBUS: Evaluating COgnitive Lateral Understanding through Multiple-choice reBUSes","date":"2024-09-06","arxiv_id":"2409.04053","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/columbus-evaluating-cognitive-lateral#ran","syntology_url":"https://syntology.ai/paper/2409.04053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.04053"}},"official":{"repos":["koen-47/columbus"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-determine-the-preferred-image","slug":"how-to-determine-the-preferred-image","title":"How to Determine the Preferred Image Distribution of a Black-Box Vision-Language Model?","date":"2024-09-03","arxiv_id":"2409.02253","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-to-determine-the-preferred-image#ran","syntology_url":"https://syntology.ai/paper/2409.02253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02253"}},"official":{"repos":["asgsaeid/cad_vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kvasir-vqa-a-text-image-pair-gi-tract-dataset","slug":"kvasir-vqa-a-text-image-pair-gi-tract-dataset","title":"Kvasir-VQA: A Text-Image Pair GI Tract Dataset","date":"2024-09-02","arxiv_id":"2409.01437","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/kvasir-vqa-a-text-image-pair-gi-tract-dataset#ran","syntology_url":"https://syntology.ai/paper/2409.01437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.01437"}},"official":{"repos":["simula/Kvasir-VQA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/evaluating-attribute-comprehension-in-large","slug":"evaluating-attribute-comprehension-in-large","title":"Evaluating Attribute Comprehension in Large Vision-Language Models","date":"2024-08-25","arxiv_id":"2408.13898","repositories_listed":1,"syntology":null},{"url":"/paper/show-o-one-single-transformer-to-unify","slug":"show-o-one-single-transformer-to-unify","title":"Show-o: One Single Transformer to Unify Multimodal Understanding and Generation","date":"2024-08-22","arxiv_id":"2408.12528","repositories_listed":1,"syntology":null},{"url":"/paper/clumo-cluster-based-modality-fusion-prompt","slug":"clumo-cluster-based-modality-fusion-prompt","title":"CluMo: Cluster-based Modality Fusion Prompt for Continual Learning in Visual Question Answering","date":"2024-08-21","arxiv_id":"2408.11742","repositories_listed":1,"syntology":null},{"url":"/paper/v-roast-a-new-dataset-for-visual-road","slug":"v-roast-a-new-dataset-for-visual-road","title":"V-RoAst: Visual Road Assessment. Can VLM be a Road Safety Assessor Using the iRAP Standard?","date":"2024-08-20","arxiv_id":"2408.10872","repositories_listed":1,"syntology":null},{"url":"/paper/teamlora-boosting-low-rank-adaptation-with","slug":"teamlora-boosting-low-rank-adaptation-with","title":"TeamLoRA: Boosting Low-Rank Adaptation with Expert Collaboration and Competition","date":"2024-08-19","arxiv_id":"2408.09856","repositories_listed":1,"syntology":null},{"url":"/paper/pa-llava-a-large-language-vision-assistant","slug":"pa-llava-a-large-language-vision-assistant","title":"PA-LLaVA: A Large Language-Vision Assistant for Human Pathology Image Understanding","date":"2024-08-18","arxiv_id":"2408.09530","repositories_listed":1,"syntology":null},{"url":"/paper/fedmeki-a-benchmark-for-scaling-medical","slug":"fedmeki-a-benchmark-for-scaling-medical","title":"FEDMEKI: A Benchmark for Scaling Medical Foundation Models via Federated Knowledge Injection","date":"2024-08-17","arxiv_id":"2408.09227","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fedmeki-a-benchmark-for-scaling-medical#ran","syntology_url":"https://syntology.ai/paper/2408.09227","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09227"}},"official":{"repos":["psudslab/FEDMEKI"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/a-survey-on-benchmarks-of-multimodal-large","slug":"a-survey-on-benchmarks-of-multimodal-large","title":"A Survey on Benchmarks of Multimodal Large Language Models","date":"2024-08-16","arxiv_id":"2408.08632","repositories_listed":1,"syntology":null},{"url":"/paper/med-pmc-medical-personalized-multi-modal","slug":"med-pmc-medical-personalized-multi-modal","title":"Med-PMC: Medical Personalized Multi-modal Consultation with a Proactive Ask-First-Observe-Next Paradigm","date":"2024-08-16","arxiv_id":"2408.08693","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/med-pmc-medical-personalized-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2408.08693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08693"}},"official":{"repos":["liuhc0428/med-pmc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-agents-as-fast-and-slow-thinkers","slug":"visual-agents-as-fast-and-slow-thinkers","title":"Visual Agents as Fast and Slow Thinkers","date":"2024-08-16","arxiv_id":"2408.08862","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-agents-as-fast-and-slow-thinkers#ran","syntology_url":"https://syntology.ai/paper/2408.08862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08862"}},"official":{"repos":["guangyans/sys2-llava"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/iiu-independent-inference-units-for-knowledge","slug":"iiu-independent-inference-units-for-knowledge","title":"IIU: Independent Inference Units for Knowledge-based Visual Question Answering","date":"2024-08-15","arxiv_id":"2408.07989","repositories_listed":1,"syntology":null},{"url":"/paper/mplug-owl3-towards-long-image-sequence","slug":"mplug-owl3-towards-long-image-sequence","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","date":"2024-08-09","arxiv_id":"2408.04840","repositories_listed":1,"syntology":null},{"url":"/paper/surgical-vqla-adversarial-contrastive","slug":"surgical-vqla-adversarial-contrastive","title":"Surgical-VQLA++: Adversarial Contrastive Learning for Calibrated Robust Visual Question-Localized Answering in Robotic Surgery","date":"2024-08-09","arxiv_id":"2408.04958","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/surgical-vqla-adversarial-contrastive#ran","syntology_url":"https://syntology.ai/paper/2408.04958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04958"}},"official":{"repos":["longbai1006/surgical-vqlaplus"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-03043","slug":"2408-03043","title":"Targeted Visual Prompting for Medical Visual Question Answering","date":"2024-08-06","arxiv_id":"2408.03043","repositories_listed":1,"syntology":null},{"url":"/paper/gmai-mmbench-a-comprehensive-multimodal","slug":"gmai-mmbench-a-comprehensive-multimodal","title":"GMAI-MMBench: A Comprehensive Multimodal Evaluation Benchmark Towards General Medical AI","date":"2024-08-06","arxiv_id":"2408.03361","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00300","slug":"2408-00300","title":"Towards Flexible Evaluation for Generative Visual Question Answering","date":"2024-08-01","arxiv_id":"2408.00300","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00765","slug":"2408-00765","title":"MM-Vet v2: A Challenging Benchmark to Evaluate Large Multimodal Models for Integrated Capabilities","date":"2024-08-01","arxiv_id":"2408.00765","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2408-00765#ran","syntology_url":"https://syntology.ai/paper/2408.00765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00765"}},"official":{"repos":["yuweihao/mm-vet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/boosting-audio-visual-question-answering-via","slug":"boosting-audio-visual-question-answering-via","title":"Boosting Audio Visual Question Answering via Key Semantic-Aware Cues","date":"2024-07-30","arxiv_id":"2407.20693","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/boosting-audio-visual-question-answering-via#ran","syntology_url":"https://syntology.ai/paper/2407.20693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.20693"}},"official":{"repos":["gewu-lab/tspm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-a-generalizable-pathology-foundation","slug":"towards-a-generalizable-pathology-foundation","title":"Towards A Generalizable Pathology Foundation Model via Unified Knowledge Distillation","date":"2024-07-26","arxiv_id":"2407.18449","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-a-generalizable-pathology-foundation#ran","syntology_url":"https://syntology.ai/paper/2407.18449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18449"}},"official":null}},{"url":"/paper/inf-llava-dual-perspective-perception-for","slug":"inf-llava-dual-perspective-perception-for","title":"INF-LLaVA: Dual-perspective Perception for High-Resolution Multimodal Large Language Model","date":"2024-07-23","arxiv_id":"2407.16198","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inf-llava-dual-perspective-perception-for#ran","syntology_url":"https://syntology.ai/paper/2407.16198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16198"}},"official":{"repos":["weihuanglin/inf-llava"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/learning-trimodal-relation-for-avqa-with","slug":"learning-trimodal-relation-for-avqa-with","title":"Learning Trimodal Relation for AVQA with Missing Modality","date":"2024-07-23","arxiv_id":"2407.16171","repositories_listed":1,"syntology":{"n":24,"n_ran":20,"n_constructed":14,"n_ran_checked":18,"n_instrument":2,"n_unverified":4,"n_honours":4,"n_violates":0,"n_no_contract":14,"n_pointer_only":3,"phrase":"20 ran (of which 14 constructed an object rather than computing a result; 18 with no instrument failure: 4 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/learning-trimodal-relation-for-avqa-with#ran","syntology_url":"https://syntology.ai/paper/2407.16171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16171"}},"official":{"repos":["visualaikhu/missing-avqa"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":14,"n_ran_no_instrument_failure":15,"n_unverified":4,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/haloquest-a-visual-hallucination-dataset-for","slug":"haloquest-a-visual-hallucination-dataset-for","title":"HaloQuest: A Visual Hallucination Dataset for Advancing Multimodal Reasoning","date":"2024-07-22","arxiv_id":"2407.15680","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-acquisition-disentanglement-for","slug":"knowledge-acquisition-disentanglement-for","title":"Knowledge Acquisition Disentanglement for Knowledge-based Visual Question Answering with Large Language Models","date":"2024-07-22","arxiv_id":"2407.15346","repositories_listed":1,"syntology":null},{"url":"/paper/mminstruct-a-high-quality-multi-modal","slug":"mminstruct-a-high-quality-multi-modal","title":"MMInstruct: A High-Quality Multi-Modal Instruction Tuning Dataset with Extensive Diversity","date":"2024-07-22","arxiv_id":"2407.15838","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mminstruct-a-high-quality-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2407.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15838"}},"official":{"repos":["yuecao0119/mminstruct"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quiil-at-t3-challenge-towards-automation-in","slug":"quiil-at-t3-challenge-towards-automation-in","title":"QuIIL at T3 challenge: Towards Automation in Life-Saving Intervention Procedures from First-Person View","date":"2024-07-18","arxiv_id":"2407.13216","repositories_listed":1,"syntology":null},{"url":"/paper/visual-haystacks-answering-harder-questions","slug":"visual-haystacks-answering-harder-questions","title":"Visual Haystacks: A Vision-Centric Needle-In-A-Haystack Benchmark","date":"2024-07-18","arxiv_id":"2407.13766","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/visual-haystacks-answering-harder-questions#ran","syntology_url":"https://syntology.ai/paper/2407.13766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13766"}},"official":{"repos":["visual-haystacks/vhs_benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/proctag-process-tagging-for-assessing-the","slug":"proctag-process-tagging-for-assessing-the","title":"ProcTag: Process Tagging for Assessing the Efficacy of Document Instruction Data","date":"2024-07-17","arxiv_id":"2407.12358","repositories_listed":1,"syntology":null},{"url":"/paper/densefusion-1m-merging-vision-experts-for","slug":"densefusion-1m-merging-vision-experts-for","title":"DenseFusion-1M: Merging Vision Experts for Comprehensive Multimodal Perception","date":"2024-07-11","arxiv_id":"2407.08303","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/densefusion-1m-merging-vision-experts-for#ran","syntology_url":"https://syntology.ai/paper/2407.08303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08303"}},"official":{"repos":["baaivision/densefusion"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-understand-layouts","slug":"large-language-models-understand-layouts","title":"Large Language Models Understand Layout","date":"2024-07-08","arxiv_id":"2407.05750","repositories_listed":1,"syntology":null},{"url":"/paper/wsi-vqa-interpreting-whole-slide-images-by","slug":"wsi-vqa-interpreting-whole-slide-images-by","title":"WSI-VQA: Interpreting Whole Slide Images by Generative Visual Question Answering","date":"2024-07-08","arxiv_id":"2407.05603","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wsi-vqa-interpreting-whole-slide-images-by#ran","syntology_url":"https://syntology.ai/paper/2407.05603","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05603"}},"official":{"repos":["cpystan/wsi-vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/minigpt-med-large-language-model-as-a-general","slug":"minigpt-med-large-language-model-as-a-general","title":"MiniGPT-Med: Large Language Model as a General Interface for Radiology Diagnosis","date":"2024-07-04","arxiv_id":"2407.04106","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minigpt-med-large-language-model-as-a-general#ran","syntology_url":"https://syntology.ai/paper/2407.04106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04106"}},"official":{"repos":["vision-cair/minigpt-med"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-to-do-if-language-models-disagree-black","slug":"what-to-do-if-language-models-disagree-black","title":"Black-box Model Ensembling for Textual and Visual Question Answering via Information Fusion","date":"2024-07-04","arxiv_id":"2407.12841","repositories_listed":1,"syntology":null},{"url":"/paper/internlm-xcomposer-2-5-a-versatile-large","slug":"internlm-xcomposer-2-5-a-versatile-large","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","date":"2024-07-03","arxiv_id":"2407.03320","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internlm-xcomposer-2-5-a-versatile-large#ran","syntology_url":"https://syntology.ai/paper/2407.03320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03320"}},"official":{"repos":["internlm/internlm-xcomposer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/visual-robustness-benchmark-for-visual","slug":"visual-robustness-benchmark-for-visual","title":"Visual Robustness Benchmark for Visual Question Answering (VQA)","date":"2024-07-03","arxiv_id":"2407.03386","repositories_listed":1,"syntology":null},{"url":"/paper/a-bounding-box-is-worth-one-token","slug":"a-bounding-box-is-worth-one-token","title":"A Bounding Box is Worth One Token: Interleaving Layout and Text in a Large Language Model for Document Understanding","date":"2024-07-02","arxiv_id":"2407.01976","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-bounding-box-is-worth-one-token#ran","syntology_url":"https://syntology.ai/paper/2407.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01976"}},"official":{"repos":["laytextllm/laytextllm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tokenpacker-efficient-visual-projector-for","slug":"tokenpacker-efficient-visual-projector-for","title":"TokenPacker: Efficient Visual Projector for Multimodal LLM","date":"2024-07-02","arxiv_id":"2407.02392","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tokenpacker-efficient-visual-projector-for#ran","syntology_url":"https://syntology.ai/paper/2407.02392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02392"}},"official":{"repos":["circleradon/tokenpacker"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cvlue-a-new-benchmark-dataset-for-chinese","slug":"cvlue-a-new-benchmark-dataset-for-chinese","title":"CVLUE: A New Benchmark Dataset for Chinese Vision-Language Understanding Evaluation","date":"2024-07-01","arxiv_id":"2407.01081","repositories_listed":1,"syntology":null},{"url":"/paper/llavolta-efficient-multi-modal-models-via","slug":"llavolta-efficient-multi-modal-models-via","title":"Efficient Large Multi-modal Models via Visual Context Compression","date":"2024-06-28","arxiv_id":"2406.20092","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llavolta-efficient-multi-modal-models-via#ran","syntology_url":"https://syntology.ai/paper/2406.20092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20092"}},"official":{"repos":["beckschen/llavolta"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stllava-med-self-training-large-language-and","slug":"stllava-med-self-training-large-language-and","title":"STLLaVA-Med: Self-Training Large Language and Vision Assistant for Medical Question-Answering","date":"2024-06-28","arxiv_id":"2406.19973","repositories_listed":1,"syntology":{"n":19,"n_ran":9,"n_constructed":4,"n_ran_checked":5,"n_instrument":4,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"9 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/stllava-med-self-training-large-language-and#ran","syntology_url":"https://syntology.ai/paper/2406.19973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19973"}},"official":{"repos":["heliossun/stllava-med"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-continual-learning-in-visual","slug":"enhancing-continual-learning-in-visual","title":"Enhancing Continual Learning in Visual Question Answering with Modality-Aware Feature Distillation","date":"2024-06-27","arxiv_id":"2406.19297","repositories_listed":1,"syntology":null},{"url":"/paper/the-illusion-of-competence-evaluating-the","slug":"the-illusion-of-competence-evaluating-the","title":"The Illusion of Competence: Evaluating the Effect of Explanations on Users' Mental Models of Visual Question Answering Systems","date":"2024-06-27","arxiv_id":"2406.19170","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-fairness-in-large-vision-language","slug":"evaluating-fairness-in-large-vision-language","title":"Evaluating Fairness in Large Vision-Language Models Across Diverse Demographic Attributes and Prompts","date":"2024-06-25","arxiv_id":"2406.17974","repositories_listed":1,"syntology":null},{"url":"/paper/mg-llava-towards-multi-granularity-visual","slug":"mg-llava-towards-multi-granularity-visual","title":"MG-LLaVA: Towards Multi-Granularity Visual Instruction Tuning","date":"2024-06-25","arxiv_id":"2406.17770","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mg-llava-towards-multi-granularity-visual#ran","syntology_url":"https://syntology.ai/paper/2406.17770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17770"}},"official":{"repos":["phoenixz810/mg-llava"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-cross-prompt-transferability-in","slug":"enhancing-cross-prompt-transferability-in","title":"Enhancing Cross-Prompt Transferability in Vision-Language Models through Contextual Injection of Target Tokens","date":"2024-06-19","arxiv_id":"2406.13294","repositories_listed":1,"syntology":null},{"url":"/paper/rationale-based-ensemble-of-multiple-qa","slug":"rationale-based-ensemble-of-multiple-qa","title":"Diversify, Rationalize, and Combine: Ensembling Multiple QA Strategies for Zero-shot Knowledge-based VQA","date":"2024-06-18","arxiv_id":"2406.12746","repositories_listed":1,"syntology":null},{"url":"/paper/trol-traversal-of-layers-for-large-language","slug":"trol-traversal-of-layers-for-large-language","title":"TroL: Traversal of Layers for Large Language and Vision Models","date":"2024-06-18","arxiv_id":"2406.12246","repositories_listed":1,"syntology":{"n":30,"n_ran":23,"n_constructed":11,"n_ran_checked":15,"n_instrument":8,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":30,"phrase":"23 ran (of which 11 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 8 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/trol-traversal-of-layers-for-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.12246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12246"}},"official":{"repos":["byungkwanlee/trol"],"state":"official (archive's flag): 23 ran","n_ran":23,"n_constructed":11,"n_ran_no_instrument_failure":15,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/mfc-bench-benchmarking-multimodal-fact","slug":"mfc-bench-benchmarking-multimodal-fact","title":"MFC-Bench: Benchmarking Multimodal Fact-Checking with Large Vision-Language Models","date":"2024-06-17","arxiv_id":"2406.11288","repositories_listed":1,"syntology":null},{"url":"/paper/mmdu-a-multi-turn-multi-image-dialog","slug":"mmdu-a-multi-turn-multi-image-dialog","title":"MMDU: A Multi-Turn Multi-Image Dialog Understanding Benchmark and Instruction-Tuning Dataset for LVLMs","date":"2024-06-17","arxiv_id":"2406.11833","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":6,"n_honours":2,"n_violates":1,"n_no_contract":1,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 1 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/mmdu-a-multi-turn-multi-image-dialog#ran","syntology_url":"https://syntology.ai/paper/2406.11833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11833"}},"official":{"repos":["liuziyu77/mmdu"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/mmneuron-discovering-neuron-level-domain","slug":"mmneuron-discovering-neuron-level-domain","title":"MMNeuron: Discovering Neuron-Level Domain-Specific Interpretation in Multimodal Large Language Model","date":"2024-06-17","arxiv_id":"2406.11193","repositories_listed":1,"syntology":null},{"url":"/paper/mixture-of-subspaces-in-low-rank-adaptation","slug":"mixture-of-subspaces-in-low-rank-adaptation","title":"Mixture-of-Subspaces in Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.11909","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mixture-of-subspaces-in-low-rank-adaptation#ran","syntology_url":"https://syntology.ai/paper/2406.11909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11909"}},"official":{"repos":["wutaiqiang/moslora"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-raw-videos-understanding-edited-videos","slug":"beyond-raw-videos-understanding-edited-videos","title":"Beyond Raw Videos: Understanding Edited Videos with Large Multimodal Model","date":"2024-06-15","arxiv_id":"2406.10484","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-models-meet-meteorology","slug":"vision-language-models-meet-meteorology","title":"Vision-Language Models Meet Meteorology: Developing Models for Extreme Weather Events Detection with Heatmaps","date":"2024-06-14","arxiv_id":"2406.09838","repositories_listed":1,"syntology":null},{"url":"/paper/explore-the-limits-of-omni-modal-pretraining","slug":"explore-the-limits-of-omni-modal-pretraining","title":"Explore the Limits of Omni-modal Pretraining at Scale","date":"2024-06-13","arxiv_id":"2406.09412","repositories_listed":1,"syntology":null},{"url":"/paper/towards-multilingual-audio-visual-question","slug":"towards-multilingual-audio-visual-question","title":"Towards Multilingual Audio-Visual Question Answering","date":"2024-06-13","arxiv_id":"2406.09156","repositories_listed":1,"syntology":null},{"url":"/paper/towards-vision-language-geo-foundation-model","slug":"towards-vision-language-geo-foundation-model","title":"Towards Vision-Language Geo-Foundation Model: A Survey","date":"2024-06-13","arxiv_id":"2406.09385","repositories_listed":1,"syntology":null},{"url":"/paper/yo-llava-your-personalized-language-and","slug":"yo-llava-your-personalized-language-and","title":"Yo'LLaVA: Your Personalized Language and Vision Assistant","date":"2024-06-13","arxiv_id":"2406.09400","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/yo-llava-your-personalized-language-and#ran","syntology_url":"https://syntology.ai/paper/2406.09400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09400"}},"official":{"repos":["WisconsinAIVision/YoLLaVA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/advancing-high-resolution-vision-language","slug":"advancing-high-resolution-vision-language","title":"Advancing High Resolution Vision-Language Models in Biomedicine","date":"2024-06-12","arxiv_id":"2406.09454","repositories_listed":1,"syntology":null},{"url":"/paper/visionllm-v2-an-end-to-end-generalist","slug":"visionllm-v2-an-end-to-end-generalist","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","date":"2024-06-12","arxiv_id":"2406.08394","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-vision-language-contrastive","slug":"benchmarking-vision-language-contrastive","title":"Benchmarking Vision-Language Contrastive Methods for Medical Representation Learning","date":"2024-06-11","arxiv_id":"2406.07450","repositories_listed":1,"syntology":null},{"url":"/paper/rs-agent-automating-remote-sensing-tasks","slug":"rs-agent-automating-remote-sensing-tasks","title":"RS-Agent: Automating Remote Sensing Tasks through Intelligent Agent","date":"2024-06-11","arxiv_id":"2406.07089","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rs-agent-automating-remote-sensing-tasks#ran","syntology_url":"https://syntology.ai/paper/2406.07089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07089"}},"official":{"repos":["intellisensing/rs-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vcr-visual-caption-restoration","slug":"vcr-visual-caption-restoration","title":"VCR: A Task for Pixel-Level Complex Reasoning in Vision Language Models via Restoring Occluded Text","date":"2024-06-10","arxiv_id":"2406.06462","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vcr-visual-caption-restoration#ran","syntology_url":"https://syntology.ai/paper/2406.06462","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06462"}},"official":{"repos":["tianyu-z/vcr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-refined-vqa-annotations-for-semi","slug":"diffusion-refined-vqa-annotations-for-semi","title":"Diffusion-Refined VQA Annotations for Semi-Supervised Gaze Following","date":"2024-06-04","arxiv_id":"2406.02774","repositories_listed":1,"syntology":null},{"url":"/paper/from-redundancy-to-relevance-enhancing","slug":"from-redundancy-to-relevance-enhancing","title":"From Redundancy to Relevance: Information Flow in LVLMs Across Reasoning Tasks","date":"2024-06-04","arxiv_id":"2406.06579","repositories_listed":1,"syntology":null},{"url":"/paper/dragonfly-multi-resolution-zoom-supercharges","slug":"dragonfly-multi-resolution-zoom-supercharges","title":"Dragonfly: Multi-Resolution Zoom-In Encoding Enhances Vision-Language Models","date":"2024-06-03","arxiv_id":"2406.00977","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dragonfly-multi-resolution-zoom-supercharges#ran","syntology_url":"https://syntology.ai/paper/2406.00977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00977"}},"official":{"repos":["togethercomputer/dragonfly"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reflection-reinforced-self-training-for","slug":"reflection-reinforced-self-training-for","title":"Re-ReST: Reflection-Reinforced Self-Training for Language Agents","date":"2024-06-03","arxiv_id":"2406.01495","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-large-vision-language-models-with","slug":"enhancing-large-vision-language-models-with","title":"Enhancing Large Vision Language Models with Self-Training on Image Comprehension","date":"2024-05-30","arxiv_id":"2405.19716","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-large-vision-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2405.19716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19716"}},"official":{"repos":["yihedeng9/stic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/instruction-guided-visual-masking","slug":"instruction-guided-visual-masking","title":"Instruction-Guided Visual Masking","date":"2024-05-30","arxiv_id":"2405.19783","repositories_listed":1,"syntology":{"n":28,"n_ran":20,"n_constructed":7,"n_ran_checked":11,"n_instrument":9,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"20 ran (of which 7 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 9 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/instruction-guided-visual-masking#ran","syntology_url":"https://syntology.ai/paper/2405.19783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19783"}},"official":{"repos":["2toinf/ivm"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":7,"n_ran_no_instrument_failure":11,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/worse-than-random-an-embarrassingly-simple","slug":"worse-than-random-an-embarrassingly-simple","title":"Worse than Random? An Embarrassingly Simple Probing Evaluation of Large Multimodal Models in Medical VQA","date":"2024-05-30","arxiv_id":"2405.20421","repositories_listed":1,"syntology":null},{"url":"/paper/reverse-image-retrieval-cues-parametric","slug":"reverse-image-retrieval-cues-parametric","title":"Reverse Image Retrieval Cues Parametric Memory in Multimodal LLMs","date":"2024-05-29","arxiv_id":"2405.18740","repositories_listed":1,"syntology":null},{"url":"/paper/convllava-hierarchical-backbones-as-visual","slug":"convllava-hierarchical-backbones-as-visual","title":"ConvLLaVA: Hierarchical Backbones as Visual Encoder for Large Multimodal Models","date":"2024-05-24","arxiv_id":"2405.15738","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/convllava-hierarchical-backbones-as-visual#ran","syntology_url":"https://syntology.ai/paper/2405.15738","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15738"}},"official":{"repos":["alibaba/conv-llava"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/meteor-mamba-based-traversal-of-rationale-for","slug":"meteor-mamba-based-traversal-of-rationale-for","title":"Meteor: Mamba-based Traversal of Rationale for Large Language and Vision Models","date":"2024-05-24","arxiv_id":"2405.15574","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/meteor-mamba-based-traversal-of-rationale-for#ran","syntology_url":"https://syntology.ai/paper/2405.15574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15574"}},"official":{"repos":["byungkwanlee/meteor"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-on-vision-language-action-models-for","slug":"a-survey-on-vision-language-action-models-for","title":"A Survey on Vision-Language-Action Models for Embodied AI","date":"2024-05-23","arxiv_id":"2405.14093","repositories_listed":1,"syntology":null},{"url":"/paper/calibrated-self-rewarding-vision-language","slug":"calibrated-self-rewarding-vision-language","title":"Calibrated Self-Rewarding Vision Language Models","date":"2024-05-23","arxiv_id":"2405.14622","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/calibrated-self-rewarding-vision-language#ran","syntology_url":"https://syntology.ai/paper/2405.14622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14622"}},"official":{"repos":["yiyangzhou/csr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-mixture-of-experts-an-auto-tuning","slug":"dynamic-mixture-of-experts-an-auto-tuning","title":"Dynamic Mixture of Experts: An Auto-Tuning Approach for Efficient Transformer Models","date":"2024-05-23","arxiv_id":"2405.14297","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-mixture-of-experts-an-auto-tuning#ran","syntology_url":"https://syntology.ai/paper/2405.14297","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14297"}},"official":{"repos":["lins-lab/dynmoe"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lova3-learning-to-visual-question-answering","slug":"lova3-learning-to-visual-question-answering","title":"LOVA3: Learning to Visual Question Answering, Asking and Assessment","date":"2024-05-23","arxiv_id":"2405.14974","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":2,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 2 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lova3-learning-to-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2405.14974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14974"}},"official":{"repos":["showlab/lova3"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/pitvqa-image-grounded-text-embedding-llm-for","slug":"pitvqa-image-grounded-text-embedding-llm-for","title":"PitVQA: Image-grounded Text Embedding LLM for Visual Question Answering in Pituitary Surgery","date":"2024-05-22","arxiv_id":"2405.13949","repositories_listed":1,"syntology":null},{"url":"/paper/dataset-and-benchmark-for-urdu-natural-scenes","slug":"dataset-and-benchmark-for-urdu-natural-scenes","title":"Dataset and Benchmark for Urdu Natural Scenes Text Detection, Recognition and Visual Question Answering","date":"2024-05-21","arxiv_id":"2405.12533","repositories_listed":1,"syntology":null},{"url":"/paper/imp-highly-capable-large-multimodal-models","slug":"imp-highly-capable-large-multimodal-models","title":"Imp: Highly Capable Large Multimodal Models for Mobile Devices","date":"2024-05-20","arxiv_id":"2405.12107","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imp-highly-capable-large-multimodal-models#ran","syntology_url":"https://syntology.ai/paper/2405.12107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.12107"}},"official":{"repos":["milvlg/imp"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mtvqa-benchmarking-multilingual-text-centric","slug":"mtvqa-benchmarking-multilingual-text-centric","title":"MTVQA: Benchmarking Multilingual Text-Centric Visual Question Answering","date":"2024-05-20","arxiv_id":"2405.11985","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mtvqa-benchmarking-multilingual-text-centric#ran","syntology_url":"https://syntology.ai/paper/2405.11985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11985"}},"official":{"repos":["bytedance/MTVQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uni-moe-scaling-unified-multimodal-llms-with","slug":"uni-moe-scaling-unified-multimodal-llms-with","title":"Uni-MoE: Scaling Unified Multimodal LLMs with Mixture of Experts","date":"2024-05-18","arxiv_id":"2405.11273","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/uni-moe-scaling-unified-multimodal-llms-with#ran","syntology_url":"https://syntology.ai/paper/2405.11273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11273"}},"official":{"repos":["hitsz-tmg/umoe-scaling-unified-multimodal-llms"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-multimodal-large-language-models-a","slug":"efficient-multimodal-large-language-models-a","title":"Efficient Multimodal Large Language Models: A Survey","date":"2024-05-17","arxiv_id":"2405.10739","repositories_listed":1,"syntology":null},{"url":"/paper/unirag-universal-retrieval-augmentation-for","slug":"unirag-universal-retrieval-augmentation-for","title":"UniRAG: Universal Retrieval Augmentation for Large Vision Language Models","date":"2024-05-16","arxiv_id":"2405.10311","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unirag-universal-retrieval-augmentation-for#ran","syntology_url":"https://syntology.ai/paper/2405.10311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.10311"}},"official":{"repos":["castorini/unirag"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/federated-document-visual-question-answering","slug":"federated-document-visual-question-answering","title":"Federated Document Visual Question Answering: A Pilot Study","date":"2024-05-10","arxiv_id":"2405.06636","repositories_listed":1,"syntology":null},{"url":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","slug":"cumo-scaling-multimodal-llm-with-co-upcycled","title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","date":"2024-05-09","arxiv_id":"2405.05949","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled#ran","syntology_url":"https://syntology.ai/paper/2405.05949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05949"}},"official":{"repos":["shi-labs/cumo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/omnidrive-a-holistic-llm-agent-framework-for","slug":"omnidrive-a-holistic-llm-agent-framework-for","title":"OmniDrive: A Holistic Vision-Language Dataset for Autonomous Driving with Counterfactual Reasoning","date":"2024-05-02","arxiv_id":"2405.01533","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/omnidrive-a-holistic-llm-agent-framework-for#ran","syntology_url":"https://syntology.ai/paper/2405.01533","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.01533"}},"official":{"repos":["nvlabs/omnidrive"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/v-flute-visual-figurative-language","slug":"v-flute-visual-figurative-language","title":"Understanding Figurative Meaning through Explainable Visual Entailment","date":"2024-05-02","arxiv_id":"2405.01474","repositories_listed":1,"syntology":null},{"url":"/paper/tablevqa-bench-a-visual-question-answering","slug":"tablevqa-bench-a-visual-question-answering","title":"TableVQA-Bench: A Visual Question Answering Benchmark on Multiple Table Domains","date":"2024-04-30","arxiv_id":"2404.19205","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tablevqa-bench-a-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2404.19205","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.19205"}},"official":{"repos":["naver-ai/tablevqabench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-page-document-visual-question-answering","slug":"multi-page-document-visual-question-answering","title":"Multi-Page Document Visual Question Answering using Self-Attention Scoring Mechanism","date":"2024-04-29","arxiv_id":"2404.19024","repositories_listed":1,"syntology":null},{"url":"/paper/how-far-are-we-to-gpt-4v-closing-the-gap-to","slug":"how-far-are-we-to-gpt-4v-closing-the-gap-to","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","date":"2024-04-25","arxiv_id":"2404.16821","repositories_listed":1,"syntology":null},{"url":"/paper/list-items-one-by-one-a-new-data-source-and","slug":"list-items-one-by-one-a-new-data-source-and","title":"List Items One by One: A New Data Source and Learning Paradigm for Multimodal LLMs","date":"2024-04-25","arxiv_id":"2404.16375","repositories_listed":1,"syntology":null},{"url":"/paper/meddr-diagnosis-guided-bootstrapping-for","slug":"meddr-diagnosis-guided-bootstrapping-for","title":"GSCo: Towards Generalizable AI in Medicine via Generalist-Specialist Collaboration","date":"2024-04-23","arxiv_id":"2404.15127","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/meddr-diagnosis-guided-bootstrapping-for#ran","syntology_url":"https://syntology.ai/paper/2404.15127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15127"}},"official":{"repos":["sunanhe/meddr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/boter-bootstrapping-knowledge-selection-and","slug":"boter-bootstrapping-knowledge-selection-and","title":"Self-Bootstrapped Visual-Language Model for Knowledge Selection and Question Answering","date":"2024-04-22","arxiv_id":"2404.13947","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/boter-bootstrapping-knowledge-selection-and#ran","syntology_url":"https://syntology.ai/paper/2404.13947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13947"}},"official":{"repos":["haodongze/self-ksel-qans"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lost-in-space-probing-fine-grained-spatial","slug":"lost-in-space-probing-fine-grained-spatial","title":"Lost in Space: Probing Fine-grained Spatial Understanding in Vision and Language Resamplers","date":"2024-04-21","arxiv_id":"2404.13594","repositories_listed":1,"syntology":null},{"url":"/paper/lapa-latent-prompt-assist-model-for-medical","slug":"lapa-latent-prompt-assist-model-for-medical","title":"LaPA: Latent Prompt Assist Model For Medical Visual Question Answering","date":"2024-04-19","arxiv_id":"2404.13039","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/lapa-latent-prompt-assist-model-for-medical#ran","syntology_url":"https://syntology.ai/paper/2404.13039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13039"}},"official":{"repos":["garygutc/lapa_model"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/look-listen-and-answer-overcoming-biases-for","slug":"look-listen-and-answer-overcoming-biases-for","title":"Look, Listen, and Answer: Overcoming Biases for Audio-Visual Question Answering","date":"2024-04-18","arxiv_id":"2404.12020","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/look-listen-and-answer-overcoming-biases-for#ran","syntology_url":"https://syntology.ai/paper/2404.12020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12020"}},"official":{"repos":["reml-group/music-avqa-r"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official"]}}}],"record_sha256":"b5a08aae8ffe1c651634d6ebbeb34c8da345156604c0db7e71a28413cee4c669","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}