{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering/papers/4","list_of":"/task/visual-question-answering","task":"Visual Question Answering (VQA)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":22,"rows_per_page":100,"rows":[301,400],"of":2167,"counts":{"archive_papers_tagged":2167,"with_a_code_link":1039,"where_syntology_ran_a_sample":359,"not_listed_spam_title":0,"listed":2167,"listed_where_code_ran":359,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":287,"every_run_a_failure_of_syntologys_instrument":72,"listed_with_a_run_with_no_instrument_failure":287,"listed_every_run_a_failure_of_syntologys_instrument":72,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering","prev":"/task/visual-question-answering/papers/3","next":"/task/visual-question-answering/papers/5","papers":[{"url":"/paper/patent-figure-classification-using-large","slug":"patent-figure-classification-using-large","title":"Patent Figure Classification using Large Vision-language Models","date":"2025-01-22","arxiv_id":"2501.12751","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-transferable-image-to-video","slug":"cross-modal-transferable-image-to-video","title":"Cross-Modal Transferable Image-to-Video Attack on Video Quality Metrics","date":"2025-01-14","arxiv_id":"2501.08415","repositories_listed":1,"syntology":null},{"url":"/paper/refocus-visual-editing-as-a-chain-of-thought","slug":"refocus-visual-editing-as-a-chain-of-thought","title":"ReFocus: Visual Editing as a Chain of Thought for Structured Image Understanding","date":"2025-01-09","arxiv_id":"2501.05452","repositories_listed":1,"syntology":null},{"url":"/paper/llava-mini-efficient-image-and-video-large","slug":"llava-mini-efficient-image-and-video-large","title":"LLaVA-Mini: Efficient Image and Video Large Multimodal Models with One Vision Token","date":"2025-01-07","arxiv_id":"2501.03895","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-mini-efficient-image-and-video-large#ran","syntology_url":"https://syntology.ai/paper/2501.03895","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.03895"}},"official":{"repos":["ictnlp/llava-mini"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automated-generation-of-challenging-multiple","slug":"automated-generation-of-challenging-multiple","title":"Automated Generation of Challenging Multiple-Choice Questions for Vision Language Model Evaluation","date":"2025-01-06","arxiv_id":"2501.03225","repositories_listed":1,"syntology":null},{"url":"/paper/redit-re-evaluating-large-visual-question","slug":"redit-re-evaluating-large-visual-question","title":"ReDiT: Re‑evaluating large visual question answering model confidence by defining input scenario Difficulty and applying Temperature mapping","date":"2025-01-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/generalizing-from-simple-to-hard-visual","slug":"generalizing-from-simple-to-hard-visual","title":"Generalizing from SIMPLE to HARD Visual Reasoning: Can We Mitigate Modality Imbalance in VLMs?","date":"2025-01-05","arxiv_id":"2501.02669","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizing-from-simple-to-hard-visual#ran","syntology_url":"https://syntology.ai/paper/2501.02669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.02669"}},"official":{"repos":["princeton-pli/vlm_s2h"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/notes-guided-mllm-reasoning-enhancing-mllm","slug":"notes-guided-mllm-reasoning-enhancing-mllm","title":"Notes-guided MLLM Reasoning: Enhancing MLLM with Knowledge and Visual Notes for Visual Question Answering","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/hallucinogen-a-benchmark-for-evaluating","slug":"hallucinogen-a-benchmark-for-evaluating","title":"HALLUCINOGEN: A Benchmark for Evaluating Object Hallucination in Large Visual-Language Models","date":"2024-12-29","arxiv_id":"2412.20622","repositories_listed":1,"syntology":null},{"url":"/paper/evalmuse-40k-a-reliable-and-fine-grained","slug":"evalmuse-40k-a-reliable-and-fine-grained","title":"EvalMuse-40K: A Reliable and Fine-Grained Benchmark with Comprehensive Human Annotations for Text-to-Image Generation Model Evaluation","date":"2024-12-24","arxiv_id":"2412.18150","repositories_listed":1,"syntology":null},{"url":"/paper/linin-logic-integrated-neural-inference","slug":"linin-logic-integrated-neural-inference","title":"LININ: Logic Integrated Neural Inference Network for Explanatory Visual Question Answering","date":"2024-12-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/instructocr-instruction-boosting-scene-text","slug":"instructocr-instruction-boosting-scene-text","title":"InstructOCR: Instruction Boosting Scene Text Spotting","date":"2024-12-20","arxiv_id":"2412.15523","repositories_listed":1,"syntology":null},{"url":"/paper/nesycoco-a-neuro-symbolic-concept-composer","slug":"nesycoco-a-neuro-symbolic-concept-composer","title":"NeSyCoCo: A Neuro-Symbolic Concept Composer for Compositional Generalization","date":"2024-12-20","arxiv_id":"2412.15588","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-hypothetical-summary-for-retrieval","slug":"multimodal-hypothetical-summary-for-retrieval","title":"Multimodal Hypothetical Summary for Retrieval-based Multi-image Question Answering","date":"2024-12-19","arxiv_id":"2412.14880","repositories_listed":1,"syntology":null},{"url":"/paper/medcot-medical-chain-of-thought-via","slug":"medcot-medical-chain-of-thought-via","title":"MedCoT: Medical Chain of Thought via Hierarchical Expert","date":"2024-12-18","arxiv_id":"2412.13736","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/medcot-medical-chain-of-thought-via#ran","syntology_url":"https://syntology.ai/paper/2412.13736","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.13736"}},"official":{"repos":["jxliu-ai/medcot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lyra-an-efficient-and-speech-centric","slug":"lyra-an-efficient-and-speech-centric","title":"Lyra: An Efficient and Speech-Centric Framework for Omni-Cognition","date":"2024-12-12","arxiv_id":"2412.09501","repositories_listed":1,"syntology":{"n":19,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":4,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/lyra-an-efficient-and-speech-centric#ran","syntology_url":"https://syntology.ai/paper/2412.09501","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09501"}},"official":{"repos":["dvlab-research/Lyra"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-a-multimodal-large-language-model","slug":"towards-a-multimodal-large-language-model","title":"Towards a Multimodal Large Language Model with Pixel-Level Insight for Biomedicine","date":"2024-12-12","arxiv_id":"2412.09278","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-a-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2412.09278","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09278"}},"official":{"repos":["shawnhuang497/medplib"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-prompt-alignment-for-text-to-image","slug":"fast-prompt-alignment-for-text-to-image","title":"Fast Prompt Alignment for Text-to-Image Generation","date":"2024-12-11","arxiv_id":"2412.08639","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fast-prompt-alignment-for-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2412.08639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08639"}},"official":{"repos":["tiktok/fast_prompt_alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/illusory-vqa-benchmarking-and-enhancing","slug":"illusory-vqa-benchmarking-and-enhancing","title":"Illusory VQA: Benchmarking and Enhancing Multimodal Models on Visual Illusions","date":"2024-12-11","arxiv_id":"2412.08169","repositories_listed":1,"syntology":null},{"url":"/paper/impact-a-large-scale-integrated-multimodal","slug":"impact-a-large-scale-integrated-multimodal","title":"IMPACT: A Large-scale Integrated Multimodal Patent Analysis and Creation Dataset for Design Patents","date":"2024-12-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mmedpo-aligning-medical-vision-language","slug":"mmedpo-aligning-medical-vision-language","title":"MMedPO: Aligning Medical Vision-Language Models with Clinical-Aware Multimodal Preference Optimization","date":"2024-12-09","arxiv_id":"2412.06141","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":8,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mmedpo-aligning-medical-vision-language#ran","syntology_url":"https://syntology.ai/paper/2412.06141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06141"}},"official":{"repos":["aiming-lab/mmedpo"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/expanding-performance-boundaries-of-open","slug":"expanding-performance-boundaries-of-open","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","date":"2024-12-06","arxiv_id":"2412.05271","repositories_listed":1,"syntology":{"n":9,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/expanding-performance-boundaries-of-open#ran","syntology_url":"https://syntology.ai/paper/2412.05271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05271"}},"official":{"repos":["opengvlab/internvl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/mammoth-vl-eliciting-multimodal-reasoning","slug":"mammoth-vl-eliciting-multimodal-reasoning","title":"MAmmoTH-VL: Eliciting Multimodal Reasoning with Instruction Tuning at Scale","date":"2024-12-06","arxiv_id":"2412.05237","repositories_listed":1,"syntology":null},{"url":"/paper/florence-vl-enhancing-vision-language-models","slug":"florence-vl-enhancing-vision-language-models","title":"Florence-VL: Enhancing Vision-Language Models with Generative Vision Encoder and Depth-Breadth Fusion","date":"2024-12-05","arxiv_id":"2412.04424","repositories_listed":1,"syntology":null},{"url":"/paper/video-quality-assessment-a-comprehensive","slug":"video-quality-assessment-a-comprehensive","title":"Video Quality Assessment: A Comprehensive Survey","date":"2024-12-04","arxiv_id":"2412.04508","repositories_listed":1,"syntology":null},{"url":"/paper/copy-move-forgery-detection-and-question","slug":"copy-move-forgery-detection-and-question","title":"Copy-Move Forgery Detection and Question Answering for Remote Sensing Image","date":"2024-12-03","arxiv_id":"2412.02575","repositories_listed":1,"syntology":null},{"url":"/paper/dlava-document-language-and-vision-assistant","slug":"dlava-document-language-and-vision-assistant","title":"DLaVA: Document Language and Vision Assistant for Answer Localization with Enhanced Interpretability and Trustworthiness","date":"2024-11-29","arxiv_id":"2412.00151","repositories_listed":1,"syntology":null},{"url":"/paper/sure-vqa-systematic-understanding-of","slug":"sure-vqa-systematic-understanding-of","title":"SURE-VQA: Systematic Understanding of Robustness Evaluation in Medical VQA Tasks","date":"2024-11-29","arxiv_id":"2411.19688","repositories_listed":1,"syntology":null},{"url":"/paper/aigv-assessor-benchmarking-and-evaluating-the","slug":"aigv-assessor-benchmarking-and-evaluating-the","title":"AIGV-Assessor: Benchmarking and Evaluating the Perceptual Quality of Text-to-Video Generation with LMM","date":"2024-11-26","arxiv_id":"2411.17221","repositories_listed":1,"syntology":null},{"url":"/paper/grounding-iqa-multimodal-language-grounding","slug":"grounding-iqa-multimodal-language-grounding","title":"Grounding-IQA: Multimodal Language Grounding Model for Image Quality Assessment","date":"2024-11-26","arxiv_id":"2411.17237","repositories_listed":1,"syntology":null},{"url":"/paper/path-rag-knowledge-guided-key-region","slug":"path-rag-knowledge-guided-key-region","title":"Path-RAG: Knowledge-Guided Key Region Retrieval for Open-ended Pathology Visual Question Answering","date":"2024-11-26","arxiv_id":"2411.17073","repositories_listed":1,"syntology":null},{"url":"/paper/zoomeye-enhancing-multimodal-llms-with-human","slug":"zoomeye-enhancing-multimodal-llms-with-human","title":"ZoomEye: Enhancing Multimodal LLMs with Human-Like Zooming Capabilities through Tree-Based Image Exploration","date":"2024-11-25","arxiv_id":"2411.16044","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zoomeye-enhancing-multimodal-llms-with-human#ran","syntology_url":"https://syntology.ai/paper/2411.16044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16044"}},"official":{"repos":["om-ai-lab/ZoomEye"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/document-haystacks-vision-language-reasoning","slug":"document-haystacks-vision-language-reasoning","title":"Document Haystacks: Vision-Language Reasoning Over Piles of 1000+ Documents","date":"2024-11-23","arxiv_id":"2411.16740","repositories_listed":1,"syntology":null},{"url":"/paper/visual-contexts-clarify-ambiguous-expressions","slug":"visual-contexts-clarify-ambiguous-expressions","title":"Visual Contexts Clarify Ambiguous Expressions: A Benchmark Dataset","date":"2024-11-21","arxiv_id":"2411.14137","repositories_listed":1,"syntology":null},{"url":"/paper/teaching-vlms-to-localize-specific-objects","slug":"teaching-vlms-to-localize-specific-objects","title":"Teaching VLMs to Localize Specific Objects from In-context Examples","date":"2024-11-20","arxiv_id":"2411.13317","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/teaching-vlms-to-localize-specific-objects#ran","syntology_url":"https://syntology.ai/paper/2411.13317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13317"}},"official":{"repos":["sivandoveh/iploc"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quantifying-preferences-of-vision-language","slug":"quantifying-preferences-of-vision-language","title":"Value-Spectrum: Quantifying Preferences of Vision-Language Models via Value Decomposition in Social Media Contexts","date":"2024-11-18","arxiv_id":"2411.11479","repositories_listed":1,"syntology":null},{"url":"/paper/awaker2-5-vl-stably-scaling-mllms-with","slug":"awaker2-5-vl-stably-scaling-mllms-with","title":"Awaker2.5-VL: Stably Scaling MLLMs with Parameter-Efficient Mixture of Experts","date":"2024-11-16","arxiv_id":"2411.10669","repositories_listed":1,"syntology":null},{"url":"/paper/sparrowvqe-visual-question-explanation-for","slug":"sparrowvqe-visual-question-explanation-for","title":"SparrowVQE: Visual Question Explanation for Course Content Understanding","date":"2024-11-12","arxiv_id":"2411.07516","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-analysis-on-spatial-reasoning","slug":"an-empirical-analysis-on-spatial-reasoning","title":"An Empirical Analysis on Spatial Reasoning Capabilities of Large Multimodal Models","date":"2024-11-09","arxiv_id":"2411.06048","repositories_listed":1,"syntology":null},{"url":"/paper/vqa-2-visual-question-answering-for-video","slug":"vqa-2-visual-question-answering-for-video","title":"VQA$^2$: Visual Question Answering for Video Quality Assessment","date":"2024-11-06","arxiv_id":"2411.03795","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-multimodal-retrieval-augmented","slug":"benchmarking-multimodal-retrieval-augmented","title":"Benchmarking Multimodal Retrieval Augmented Generation with Dynamic VQA Dataset and Self-adaptive Planning Agent","date":"2024-11-05","arxiv_id":"2411.02937","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-multimodal-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2411.02937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02937"}},"official":{"repos":["alibaba-nlp/omnisearch"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-vision-language-model-unlearning","slug":"benchmarking-vision-language-model-unlearning","title":"Benchmarking Vision Language Model Unlearning via Fictitious Facial Identity Dataset","date":"2024-11-05","arxiv_id":"2411.03554","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-vision-language-model-unlearning#ran","syntology_url":"https://syntology.ai/paper/2411.03554","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03554"}},"official":{"repos":["safolab-wisc/fiubench"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/right-this-way-can-vlms-guide-us-to-see-more","slug":"right-this-way-can-vlms-guide-us-to-see-more","title":"Right this way: Can VLMs Guide Us to See More to Answer Questions?","date":"2024-11-01","arxiv_id":"2411.00394","repositories_listed":1,"syntology":{"n":17,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":13,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/right-this-way-can-vlms-guide-us-to-see-more#ran","syntology_url":"https://syntology.ai/paper/2411.00394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00394"}},"official":{"repos":["LeoLee7/Directional_guidance"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":13,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/are-vlms-really-blind","slug":"are-vlms-really-blind","title":"Are VLMs Really Blind","date":"2024-10-29","arxiv_id":"2410.22029","repositories_listed":1,"syntology":null},{"url":"/paper/autobench-v-can-large-vision-language-models","slug":"autobench-v-can-large-vision-language-models","title":"AutoBench-V: Can Large Vision-Language Models Benchmark Themselves?","date":"2024-10-28","arxiv_id":"2410.21259","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autobench-v-can-large-vision-language-models#ran","syntology_url":"https://syntology.ai/paper/2410.21259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21259"}},"official":{"repos":["wad3birch/AutoBench-V"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/few-shot-multimodal-explanation-for-visual","slug":"few-shot-multimodal-explanation-for-visual","title":"Few-Shot Multimodal Explanation for Visual Question Answering","date":"2024-10-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/frontiers-in-intelligent-colonoscopy","slug":"frontiers-in-intelligent-colonoscopy","title":"Frontiers in Intelligent Colonoscopy","date":"2024-10-22","arxiv_id":"2410.17241","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/frontiers-in-intelligent-colonoscopy#ran","syntology_url":"https://syntology.ai/paper/2410.17241","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17241"}},"official":{"repos":["ai4colonoscopy/intelliscope"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/progressive-compositionality-in-text-to-image","slug":"progressive-compositionality-in-text-to-image","title":"Progressive Compositionality In Text-to-Image Generative Models","date":"2024-10-22","arxiv_id":"2410.16719","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/progressive-compositionality-in-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2410.16719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16719"}},"official":{"repos":["evansh666/evogen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/griffon-g-bridging-vision-language-and-vision","slug":"griffon-g-bridging-vision-language-and-vision","title":"Griffon-G: Bridging Vision-Language and Vision-Centric Tasks via Large Multimodal Models","date":"2024-10-21","arxiv_id":"2410.16163","repositories_listed":1,"syntology":null},{"url":"/paper/semihvision-enhancing-medical-multimodal","slug":"semihvision-enhancing-medical-multimodal","title":"SemiHVision: Enhancing Medical Multimodal Models with a Semi-Human Annotated Dataset and Fine-Tuned Instruction Generation","date":"2024-10-19","arxiv_id":"2410.14948","repositories_listed":1,"syntology":null},{"url":"/paper/viconsformer-constituting-meaningful-phrases","slug":"viconsformer-constituting-meaningful-phrases","title":"ViConsFormer: Constituting Meaningful Phrases of Scene Texts using Transformer-based Method in Vietnamese Text-based Visual Question Answering","date":"2024-10-18","arxiv_id":"2410.14132","repositories_listed":1,"syntology":null},{"url":"/paper/actioncomet-a-zero-shot-approach-to-learn","slug":"actioncomet-a-zero-shot-approach-to-learn","title":"ActionCOMET: A Zero-shot Approach to Learn Image-specific Commonsense Concepts about Actions","date":"2024-10-17","arxiv_id":"2410.13662","repositories_listed":1,"syntology":null},{"url":"/paper/help-me-identify-is-an-llm-vqa-system-all-we","slug":"help-me-identify-is-an-llm-vqa-system-all-we","title":"Help Me Identify: Is an LLM+VQA System All We Need to Identify Visual Concepts?","date":"2024-10-17","arxiv_id":"2410.13651","repositories_listed":1,"syntology":null},{"url":"/paper/mmed-rag-versatile-multimodal-rag-system-for","slug":"mmed-rag-versatile-multimodal-rag-system-for","title":"MMed-RAG: Versatile Multimodal RAG System for Medical Vision Language Models","date":"2024-10-16","arxiv_id":"2410.13085","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mmed-rag-versatile-multimodal-rag-system-for#ran","syntology_url":"https://syntology.ai/paper/2410.13085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13085"}},"official":{"repos":["richard-peng-xia/mmed-rag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vividmed-vision-language-model-with-versatile","slug":"vividmed-vision-language-model-with-versatile","title":"VividMed: Vision Language Model with Versatile Visual Grounding for Medicine","date":"2024-10-16","arxiv_id":"2410.12694","repositories_listed":1,"syntology":null},{"url":"/paper/worldcuisines-a-massive-scale-benchmark-for","slug":"worldcuisines-a-massive-scale-benchmark-for","title":"WorldCuisines: A Massive-Scale Benchmark for Multilingual and Multicultural Visual Question Answering on Global Cuisines","date":"2024-10-16","arxiv_id":"2410.12705","repositories_listed":1,"syntology":{"n":17,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/worldcuisines-a-massive-scale-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2410.12705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12705"}},"official":{"repos":["worldcuisines/worldcuisines"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/difficult-task-yes-but-simple-task-no","slug":"difficult-task-yes-but-simple-task-no","title":"Difficult Task Yes but Simple Task No: Unveiling the Laziness in Multimodal LLMs","date":"2024-10-15","arxiv_id":"2410.11437","repositories_listed":1,"syntology":null},{"url":"/paper/livexiv-a-multi-modal-live-benchmark-based-on","slug":"livexiv-a-multi-modal-live-benchmark-based-on","title":"LiveXiv -- A Multi-Modal Live Benchmark Based on Arxiv Papers Content","date":"2024-10-14","arxiv_id":"2410.10783","repositories_listed":1,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":16,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/livexiv-a-multi-modal-live-benchmark-based-on#ran","syntology_url":"https://syntology.ai/paper/2410.10783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10783"}},"official":{"repos":["nimrodshabtay/livexiv"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/declarative-knowledge-distillation-from-large","slug":"declarative-knowledge-distillation-from-large","title":"Declarative Knowledge Distillation from Large Language Models for Visual Question Answering Datasets","date":"2024-10-12","arxiv_id":"2410.09428","repositories_listed":1,"syntology":null},{"url":"/paper/skipping-computations-in-multimodal-llms","slug":"skipping-computations-in-multimodal-llms","title":"Skipping Computations in Multimodal LLMs","date":"2024-10-12","arxiv_id":"2410.09454","repositories_listed":1,"syntology":null},{"url":"/paper/dataenvgym-data-generation-agents-in-teacher","slug":"dataenvgym-data-generation-agents-in-teacher","title":"DataEnvGym: Data Generation Agents in Teacher Environments with Student Feedback","date":"2024-10-08","arxiv_id":"2410.06215","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dataenvgym-data-generation-agents-in-teacher#ran","syntology_url":"https://syntology.ai/paper/2410.06215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06215"}},"official":{"repos":["codezakh/dataenvgym"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ervqa-a-dataset-to-benchmark-the-readiness-of","slug":"ervqa-a-dataset-to-benchmark-the-readiness-of","title":"ERVQA: A Dataset to Benchmark the Readiness of Large Vision Language Models in Hospital Environments","date":"2024-10-08","arxiv_id":"2410.06420","repositories_listed":1,"syntology":null},{"url":"/paper/actiview-evaluating-active-perception-ability","slug":"actiview-evaluating-active-perception-ability","title":"ActiView: Evaluating Active Perception Ability for Multimodal Large Language Models","date":"2024-10-07","arxiv_id":"2410.04659","repositories_listed":1,"syntology":null},{"url":"/paper/mc-cot-a-modular-collaborative-cot-framework","slug":"mc-cot-a-modular-collaborative-cot-framework","title":"MC-CoT: A Modular Collaborative CoT Framework for Zero-shot Medical-VQA with LLM and MLLM Integration","date":"2024-10-06","arxiv_id":"2410.04521","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":15,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mc-cot-a-modular-collaborative-cot-framework#ran","syntology_url":"https://syntology.ai/paper/2410.04521","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04521"}},"official":{"repos":["thomaswei-cn/MC-CoT"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tubench-benchmarking-large-vision-language","slug":"tubench-benchmarking-large-vision-language","title":"TUBench: Benchmarking Large Vision-Language Models on Trustworthiness with Unanswerable Questions","date":"2024-10-05","arxiv_id":"2410.04107","repositories_listed":1,"syntology":null},{"url":"/paper/badcm-invisible-backdoor-attack-against-cross","slug":"badcm-invisible-backdoor-attack-against-cross","title":"BadCM: Invisible Backdoor Attack Against Cross-Modal Learning","date":"2024-10-03","arxiv_id":"2410.02182","repositories_listed":1,"syntology":null},{"url":"/paper/a-hitchhikers-guide-to-fine-grained-face","slug":"a-hitchhikers-guide-to-fine-grained-face","title":"A Hitchhikers Guide to Fine-Grained Face Forgery Detection Using Common Sense Reasoning","date":"2024-10-01","arxiv_id":"2410.00485","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-hitchhikers-guide-to-fine-grained-face#ran","syntology_url":"https://syntology.ai/paper/2410.00485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00485"}},"official":{"repos":["NickyFot/HitchhikersGuide"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/babelbench-an-omni-benchmark-for-code-driven","slug":"babelbench-an-omni-benchmark-for-code-driven","title":"BabelBench: An Omni Benchmark for Code-Driven Analysis of Multimodal and Multistructured Data","date":"2024-10-01","arxiv_id":"2410.00773","repositories_listed":1,"syntology":null},{"url":"/paper/unleashing-the-potentials-of-likelihood","slug":"unleashing-the-potentials-of-likelihood","title":"Unleashing the Potentials of Likelihood Composition for Multi-modal Language Models","date":"2024-10-01","arxiv_id":"2410.00363","repositories_listed":1,"syntology":null},{"url":"/paper/t2vs-meet-vlms-a-scalable-multimodal-dataset","slug":"t2vs-meet-vlms-a-scalable-multimodal-dataset","title":"T2Vs Meet VLMs: A Scalable Multimodal Dataset for Visual Harmfulness Recognition","date":"2024-09-29","arxiv_id":"2409.19734","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/t2vs-meet-vlms-a-scalable-multimodal-dataset#ran","syntology_url":"https://syntology.ai/paper/2409.19734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.19734"}},"official":{"repos":["nctu-eva-lab/vhd11k"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/a-unified-hallucination-mitigation-framework","slug":"a-unified-hallucination-mitigation-framework","title":"A Unified Hallucination Mitigation Framework for Large Vision-Language Models","date":"2024-09-24","arxiv_id":"2409.16494","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-video-quality-assessment-from-the","slug":"revisiting-video-quality-assessment-from-the","title":"Revisiting Video Quality Assessment from the Perspective of Generalization","date":"2024-09-23","arxiv_id":"2409.14847","repositories_listed":1,"syntology":null},{"url":"/paper/towards-efficient-and-robust-vqa-nle-data","slug":"towards-efficient-and-robust-vqa-nle-data","title":"Towards Efficient and Robust VQA-NLE Data Generation with Large Vision-Language Models","date":"2024-09-23","arxiv_id":"2409.14785","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-image-hallucination-in-text-to","slug":"evaluating-image-hallucination-in-text-to","title":"Evaluating Image Hallucination in Text-to-Image Generation with Question-Answering","date":"2024-09-19","arxiv_id":"2409.12784","repositories_listed":1,"syntology":null},{"url":"/paper/journeybench-a-challenging-one-stop-vision","slug":"journeybench-a-challenging-one-stop-vision","title":"JourneyBench: A Challenging One-Stop Vision-Language Understanding Benchmark of Generated Images","date":"2024-09-19","arxiv_id":"2409.12953","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/journeybench-a-challenging-one-stop-vision#ran","syntology_url":"https://syntology.ai/paper/2409.12953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12953"}},"official":{"repos":["journeybench/journeybench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cast-cross-modal-alignment-similarity-test","slug":"cast-cross-modal-alignment-similarity-test","title":"CAST: Cross-modal Alignment Similarity Test for Vision Language Models","date":"2024-09-17","arxiv_id":"2409.11007","repositories_listed":1,"syntology":null},{"url":"/paper/less-is-more-a-simple-yet-effective-token","slug":"less-is-more-a-simple-yet-effective-token","title":"Less is More: A Simple yet Effective Token Reduction Method for Efficient Multi-modal LLMs","date":"2024-09-17","arxiv_id":"2409.10994","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/less-is-more-a-simple-yet-effective-token#ran","syntology_url":"https://syntology.ai/paper/2409.10994","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.10994"}},"official":{"repos":["freedomintelligence/trim"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/guiding-vision-language-model-selection-for","slug":"guiding-vision-language-model-selection-for","title":"Guiding Vision-Language Model Selection for Visual Question-Answering Across Tasks, Domains, and Knowledge Types","date":"2024-09-14","arxiv_id":"2409.09269","repositories_listed":1,"syntology":null},{"url":"/paper/columbus-evaluating-cognitive-lateral","slug":"columbus-evaluating-cognitive-lateral","title":"COLUMBUS: Evaluating COgnitive Lateral Understanding through Multiple-choice reBUSes","date":"2024-09-06","arxiv_id":"2409.04053","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/columbus-evaluating-cognitive-lateral#ran","syntology_url":"https://syntology.ai/paper/2409.04053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.04053"}},"official":{"repos":["koen-47/columbus"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-determine-the-preferred-image","slug":"how-to-determine-the-preferred-image","title":"How to Determine the Preferred Image Distribution of a Black-Box Vision-Language Model?","date":"2024-09-03","arxiv_id":"2409.02253","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-to-determine-the-preferred-image#ran","syntology_url":"https://syntology.ai/paper/2409.02253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02253"}},"official":{"repos":["asgsaeid/cad_vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kvasir-vqa-a-text-image-pair-gi-tract-dataset","slug":"kvasir-vqa-a-text-image-pair-gi-tract-dataset","title":"Kvasir-VQA: A Text-Image Pair GI Tract Dataset","date":"2024-09-02","arxiv_id":"2409.01437","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/kvasir-vqa-a-text-image-pair-gi-tract-dataset#ran","syntology_url":"https://syntology.ai/paper/2409.01437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.01437"}},"official":{"repos":["simula/Kvasir-VQA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/gradbias-unveiling-word-influence-on-bias-in","slug":"gradbias-unveiling-word-influence-on-bias-in","title":"GradBias: Unveiling Word Influence on Bias in Text-to-Image Generative Models","date":"2024-08-29","arxiv_id":"2408.16700","repositories_listed":1,"syntology":null},{"url":"/paper/aim-2024-challenge-on-compressed-video","slug":"aim-2024-challenge-on-compressed-video","title":"AIM 2024 Challenge on Compressed Video Quality Assessment: Methods and Results","date":"2024-08-21","arxiv_id":"2408.11982","repositories_listed":1,"syntology":null},{"url":"/paper/v-roast-a-new-dataset-for-visual-road","slug":"v-roast-a-new-dataset-for-visual-road","title":"V-RoAst: Visual Road Assessment. Can VLM be a Road Safety Assessor Using the iRAP Standard?","date":"2024-08-20","arxiv_id":"2408.10872","repositories_listed":1,"syntology":null},{"url":"/paper/ffaa-multimodal-large-language-model-based","slug":"ffaa-multimodal-large-language-model-based","title":"FFAA: Multimodal Large Language Model based Explainable Open-World Face Forgery Analysis Assistant","date":"2024-08-19","arxiv_id":"2408.10072","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/ffaa-multimodal-large-language-model-based#ran","syntology_url":"https://syntology.ai/paper/2408.10072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.10072"}},"official":{"repos":["thu-huangzc/FFAA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/pa-llava-a-large-language-vision-assistant","slug":"pa-llava-a-large-language-vision-assistant","title":"PA-LLaVA: A Large Language-Vision Assistant for Human Pathology Image Understanding","date":"2024-08-18","arxiv_id":"2408.09530","repositories_listed":1,"syntology":null},{"url":"/paper/med-pmc-medical-personalized-multi-modal","slug":"med-pmc-medical-personalized-multi-modal","title":"Med-PMC: Medical Personalized Multi-modal Consultation with a Proactive Ask-First-Observe-Next Paradigm","date":"2024-08-16","arxiv_id":"2408.08693","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/med-pmc-medical-personalized-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2408.08693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08693"}},"official":{"repos":["liuhc0428/med-pmc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-agents-as-fast-and-slow-thinkers","slug":"visual-agents-as-fast-and-slow-thinkers","title":"Visual Agents as Fast and Slow Thinkers","date":"2024-08-16","arxiv_id":"2408.08862","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-agents-as-fast-and-slow-thinkers#ran","syntology_url":"https://syntology.ai/paper/2408.08862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08862"}},"official":{"repos":["guangyans/sys2-llava"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mplug-owl3-towards-long-image-sequence","slug":"mplug-owl3-towards-long-image-sequence","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","date":"2024-08-09","arxiv_id":"2408.04840","repositories_listed":1,"syntology":null},{"url":"/paper/surgical-vqla-adversarial-contrastive","slug":"surgical-vqla-adversarial-contrastive","title":"Surgical-VQLA++: Adversarial Contrastive Learning for Calibrated Robust Visual Question-Localized Answering in Robotic Surgery","date":"2024-08-09","arxiv_id":"2408.04958","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/surgical-vqla-adversarial-contrastive#ran","syntology_url":"https://syntology.ai/paper/2408.04958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04958"}},"official":{"repos":["longbai1006/surgical-vqlaplus"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-02900","slug":"2408-02900","title":"MedTrinity-25M: A Large-scale Multimodal Dataset with Multigranular Annotations for Medicine","date":"2024-08-06","arxiv_id":"2408.02900","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-02900#ran","syntology_url":"https://syntology.ai/paper/2408.02900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.02900"}},"official":{"repos":["UCSC-VLAA/MedTrinity-25M"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-03043","slug":"2408-03043","title":"Targeted Visual Prompting for Medical Visual Question Answering","date":"2024-08-06","arxiv_id":"2408.03043","repositories_listed":1,"syntology":null},{"url":"/paper/gmai-mmbench-a-comprehensive-multimodal","slug":"gmai-mmbench-a-comprehensive-multimodal","title":"GMAI-MMBench: A Comprehensive Multimodal Evaluation Benchmark Towards General Medical AI","date":"2024-08-06","arxiv_id":"2408.03361","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00300","slug":"2408-00300","title":"Towards Flexible Evaluation for Generative Visual Question Answering","date":"2024-08-01","arxiv_id":"2408.00300","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-low-rank-learning-bella-a-practical","slug":"bayesian-low-rank-learning-bella-a-practical","title":"Bayesian Low-Rank LeArning (Bella): A Practical Approach to Bayesian Neural Networks","date":"2024-07-30","arxiv_id":"2407.20891","repositories_listed":1,"syntology":null},{"url":"/paper/multi-label-cluster-discrimination-for-visual","slug":"multi-label-cluster-discrimination-for-visual","title":"Multi-label Cluster Discrimination for Visual Representation Learning","date":"2024-07-24","arxiv_id":"2407.17331","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":7,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 7 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 7 samples that ran constructed an object rather than computing a result","sample_list":"/paper/multi-label-cluster-discrimination-for-visual#ran","syntology_url":"https://syntology.ai/paper/2407.17331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17331"}},"official":{"repos":["deepglint/unicom"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":7,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/quiil-at-t3-challenge-towards-automation-in","slug":"quiil-at-t3-challenge-towards-automation-in","title":"QuIIL at T3 challenge: Towards Automation in Life-Saving Intervention Procedures from First-Person View","date":"2024-07-18","arxiv_id":"2407.13216","repositories_listed":1,"syntology":null},{"url":"/paper/visual-haystacks-answering-harder-questions","slug":"visual-haystacks-answering-harder-questions","title":"Visual Haystacks: A Vision-Centric Needle-In-A-Haystack Benchmark","date":"2024-07-18","arxiv_id":"2407.13766","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/visual-haystacks-answering-harder-questions#ran","syntology_url":"https://syntology.ai/paper/2407.13766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13766"}},"official":{"repos":["visual-haystacks/vhs_benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/proctag-process-tagging-for-assessing-the","slug":"proctag-process-tagging-for-assessing-the","title":"ProcTag: Process Tagging for Assessing the Efficacy of Document Instruction Data","date":"2024-07-17","arxiv_id":"2407.12358","repositories_listed":1,"syntology":null},{"url":"/paper/relax-vqa-residual-fragment-and-layer-stack","slug":"relax-vqa-residual-fragment-and-layer-stack","title":"ReLaX-VQA: Residual Fragment and Layer Stack Extraction for Enhancing Video Quality Assessment","date":"2024-07-16","arxiv_id":"2407.11496","repositories_listed":1,"syntology":null}],"record_sha256":"a34b96bcf6b769e33b9fb00f0f5a499d4f6b0260a4a95f21c1a8b22d650da4cb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}