{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering/papers/11","list_of":"/task/visual-question-answering","task":"Visual Question Answering (VQA)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":22,"rows_per_page":100,"rows":[1001,1100],"of":2167,"counts":{"archive_papers_tagged":2167,"with_a_code_link":1039,"where_syntology_ran_a_sample":359,"not_listed_spam_title":0,"listed":2167,"listed_where_code_ran":359,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":287,"every_run_a_failure_of_syntologys_instrument":72,"listed_with_a_run_with_no_instrument_failure":287,"listed_every_run_a_failure_of_syntologys_instrument":72,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering","prev":"/task/visual-question-answering/papers/10","next":"/task/visual-question-answering/papers/12","papers":[{"url":"/paper/vizwiz-grand-challenge-answering-visual","slug":"vizwiz-grand-challenge-answering-visual","title":"VizWiz Grand Challenge: Answering Visual Questions from Blind People","date":"2018-02-22","arxiv_id":"1802.08218","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-count-objects-in-natural-images","slug":"learning-to-count-objects-in-natural-images","title":"Learning to Count Objects in Natural Images for Visual Question Answering","date":"2018-02-15","arxiv_id":"1802.05766","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-explanations-justifying-decisions","slug":"multimodal-explanations-justifying-decisions","title":"Multimodal Explanations: Justifying Decisions and Pointing to the Evidence","date":"2018-02-15","arxiv_id":"1802.08129","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-explanations-justifying-decisions#ran","syntology_url":"https://syntology.ai/paper/1802.08129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.08129"}},"official":{"repos":["Seth-Park/MultimodalExplanations"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-recurrent-attention-units-for-visual","slug":"dual-recurrent-attention-units-for-visual","title":"Dual Recurrent Attention Units for Visual Question Answering","date":"2018-02-01","arxiv_id":"1802.00209","repositories_listed":1,"syntology":null},{"url":"/paper/game-of-sketches-deep-recurrent-models-of","slug":"game-of-sketches-deep-recurrent-models-of","title":"Game of Sketches: Deep Recurrent Models of Pictionary-style Word Guessing","date":"2018-01-29","arxiv_id":"1801.09356","repositories_listed":1,"syntology":null},{"url":"/paper/dvqa-understanding-data-visualizations-via","slug":"dvqa-understanding-data-visualizations-via","title":"DVQA: Understanding Data Visualizations via Question Answering","date":"2018-01-24","arxiv_id":"1801.08163","repositories_listed":1,"syntology":null},{"url":"/paper/structured-triplet-learning-with-pos-tag","slug":"structured-triplet-learning-with-pos-tag","title":"Structured Triplet Learning with POS-tag Guided Attention for Visual Question Answering","date":"2018-01-24","arxiv_id":"1801.07853","repositories_listed":1,"syntology":null},{"url":"/paper/iqa-visual-question-answering-in-interactive","slug":"iqa-visual-question-answering-in-interactive","title":"IQA: Visual Question Answering in Interactive Environments","date":"2017-12-09","arxiv_id":"1712.03316","repositories_listed":1,"syntology":null},{"url":"/paper/dont-just-assume-look-and-answer-overcoming","slug":"dont-just-assume-look-and-answer-overcoming","title":"Don't Just Assume; Look and Answer: Overcoming Priors for Visual Question Answering","date":"2017-12-01","arxiv_id":"1712.00377","repositories_listed":1,"syntology":null},{"url":"/paper/locally-smoothed-neural-networks","slug":"locally-smoothed-neural-networks","title":"Locally Smoothed Neural Networks","date":"2017-11-22","arxiv_id":"1711.08132","repositories_listed":1,"syntology":null},{"url":"/paper/co-attending-free-form-regions-and-detections","slug":"co-attending-free-form-regions-and-detections","title":"Co-attending Free-form Regions and Detections with Multi-modal Multiplicative Feature Embedding for Visual Question Answering","date":"2017-11-18","arxiv_id":"1711.06794","repositories_listed":1,"syntology":null},{"url":"/paper/co-attending-regions-and-detections-with","slug":"co-attending-regions-and-detections-with","title":"Co-attending Regions and Detections with Multi-modal Multiplicative Embedding for VQA","date":"2017-11-18","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/high-order-attention-models-for-visual","slug":"high-order-attention-models-for-visual","title":"High-Order Attention Models for Visual Question Answering","date":"2017-11-12","arxiv_id":"1711.04323","repositories_listed":1,"syntology":null},{"url":"/paper/active-learning-for-visual-question-answering","slug":"active-learning-for-visual-question-answering","title":"Active Learning for Visual Question Answering: An Empirical Study","date":"2017-11-06","arxiv_id":"1711.01732","repositories_listed":1,"syntology":null},{"url":"/paper/figureqa-an-annotated-figure-dataset-for","slug":"figureqa-an-annotated-figure-dataset-for","title":"FigureQA: An Annotated Figure Dataset for Visual Reasoning","date":"2017-10-19","arxiv_id":"1710.07300","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":12,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/figureqa-an-annotated-figure-dataset-for#ran","syntology_url":"https://syntology.ai/paper/1710.07300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.07300"}},"official":{"repos":["vmichals/FigureQA-baseline"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/vqs-linking-segmentations-to-questions-and","slug":"vqs-linking-segmentations-to-questions-and","title":"VQS: Linking Segmentations to Questions and Answers for Supervised Attention in VQA and Question-Focused Semantic Segmentation","date":"2017-08-15","arxiv_id":"1708.04686","repositories_listed":1,"syntology":null},{"url":"/paper/structured-attentions-for-visual-question","slug":"structured-attentions-for-visual-question","title":"Structured Attentions for Visual Question Answering","date":"2017-08-07","arxiv_id":"1708.02071","repositories_listed":1,"syntology":null},{"url":"/paper/effective-approaches-to-batch-parallelization","slug":"effective-approaches-to-batch-parallelization","title":"Effective Approaches to Batch Parallelization for Dynamic Neural Network Architectures","date":"2017-07-08","arxiv_id":"1707.02402","repositories_listed":1,"syntology":null},{"url":"/paper/learning-convolutional-text-representations","slug":"learning-convolutional-text-representations","title":"Learning Convolutional Text Representations for Visual Question Answering","date":"2017-05-18","arxiv_id":"1705.06824","repositories_listed":1,"syntology":null},{"url":"/paper/speech-based-visual-question-answering","slug":"speech-based-visual-question-answering","title":"Speech-Based Visual Question Answering","date":"2017-05-01","arxiv_id":"1705.00464","repositories_listed":1,"syntology":null},{"url":"/paper/the-promise-of-premise-harnessing-question","slug":"the-promise-of-premise-harnessing-question","title":"The Promise of Premise: Harnessing Question Premises in Visual Question Answering","date":"2017-05-01","arxiv_id":"1705.00601","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-reason-end-to-end-module-networks","slug":"learning-to-reason-end-to-end-module-networks","title":"Learning to Reason: End-to-End Module Networks for Visual Question Answering","date":"2017-04-18","arxiv_id":"1704.05526","repositories_listed":1,"syntology":null},{"url":"/paper/whats-in-a-question-using-visual-questions-as","slug":"whats-in-a-question-using-visual-questions-as","title":"What's in a Question: Using Visual Questions as a Form of Supervision","date":"2017-04-12","arxiv_id":"1704.03895","repositories_listed":1,"syntology":null},{"url":"/paper/vibiknet-visual-bidirectional-kernelized","slug":"vibiknet-visual-bidirectional-kernelized","title":"VIBIKNet: Visual Bidirectional Kernelized Network for Visual Question Answering","date":"2016-12-12","arxiv_id":"1612.03628","repositories_listed":1,"syntology":null},{"url":"/paper/open-ended-visual-question-answering","slug":"open-ended-visual-question-answering","title":"Open-Ended Visual Question-Answering","date":"2016-10-09","arxiv_id":"1610.02692","repositories_listed":1,"syntology":null},{"url":"/paper/visual-question-answering-datasets-algorithms","slug":"visual-question-answering-datasets-algorithms","title":"Visual Question Answering: Datasets, Algorithms, and Future Challenges","date":"2016-10-05","arxiv_id":"1610.01465","repositories_listed":1,"syntology":null},{"url":"/paper/tutorial-on-answering-questions-about-images","slug":"tutorial-on-answering-questions-about-images","title":"Tutorial on Answering Questions about Images with Deep Learning","date":"2016-10-04","arxiv_id":"1610.01076","repositories_listed":1,"syntology":null},{"url":"/paper/visual-question-answering-a-survey-of-methods","slug":"visual-question-answering-a-survey-of-methods","title":"Visual Question Answering: A Survey of Methods and Datasets","date":"2016-07-20","arxiv_id":"1607.05910","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-the-behavior-of-visual-question","slug":"analyzing-the-behavior-of-visual-question","title":"Analyzing the Behavior of Visual Question Answering Models","date":"2016-06-23","arxiv_id":"1606.07356","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/analyzing-the-behavior-of-visual-question#ran","syntology_url":"https://syntology.ai/paper/1606.07356","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.07356"}},"official":{"repos":["akirafukui/vqa-mcb"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-residual-learning-for-visual-qa","slug":"multimodal-residual-learning-for-visual-qa","title":"Multimodal Residual Learning for Visual QA","date":"2016-06-05","arxiv_id":"1606.01455","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-instance-segmentation-with","slug":"end-to-end-instance-segmentation-with","title":"End-to-End Instance Segmentation with Recurrent Attention","date":"2016-05-30","arxiv_id":"1605.09410","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/end-to-end-instance-segmentation-with#ran","syntology_url":"https://syntology.ai/paper/1605.09410","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1605.09410"}},"official":null}},{"url":"/paper/ask-your-neurons-a-deep-learning-approach-to","slug":"ask-your-neurons-a-deep-learning-approach-to","title":"Ask Your Neurons: A Deep Learning Approach to Visual Question Answering","date":"2016-05-09","arxiv_id":"1605.02697","repositories_listed":1,"syntology":null},{"url":"/paper/counting-everyday-objects-in-everyday-scenes","slug":"counting-everyday-objects-in-everyday-scenes","title":"Counting Everyday Objects in Everyday Scenes","date":"2016-04-12","arxiv_id":"1604.03505","repositories_listed":1,"syntology":null},{"url":"/paper/a-diagram-is-worth-a-dozen-images","slug":"a-diagram-is-worth-a-dozen-images","title":"A Diagram Is Worth A Dozen Images","date":"2016-03-24","arxiv_id":"1603.07396","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-diagram-is-worth-a-dozen-images#ran","syntology_url":"https://syntology.ai/paper/1603.07396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1603.07396"}},"official":{"repos":["allenai/dqa-net"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/image-question-answering-using-convolutional","slug":"image-question-answering-using-convolutional","title":"Image Question Answering using Convolutional Neural Network with Dynamic Parameter Prediction","date":"2015-11-18","arxiv_id":"1511.05756","repositories_listed":1,"syntology":null},{"url":"/paper/ask-attend-and-answer-exploring-question","slug":"ask-attend-and-answer-exploring-question","title":"Ask, Attend and Answer: Exploring Question-Guided Spatial Attention for Visual Question Answering","date":"2015-11-17","arxiv_id":"1511.05234","repositories_listed":1,"syntology":null},{"url":"/paper/what-value-do-explicit-high-level-concepts","slug":"what-value-do-explicit-high-level-concepts","title":"What value do explicit high level concepts have in vision to language problems?","date":"2015-06-03","arxiv_id":"1506.01144","repositories_listed":1,"syntology":null},{"url":"/paper/are-you-talking-to-a-machine-dataset-and","slug":"are-you-talking-to-a-machine-dataset-and","title":"Are You Talking to a Machine? Dataset and Methods for Multilingual Image Question Answering","date":"2015-05-21","arxiv_id":"1505.05612","repositories_listed":1,"syntology":null},{"url":"/paper/blind-prediction-of-natural-video-quality","slug":"blind-prediction-of-natural-video-quality","title":"Blind Prediction of Natural Video Quality","date":"2014-01-09","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":null,"slug":"visionthink-smart-and-efficient-vision","title":"VisionThink: Smart and Efficient Vision Language Model via Reinforcement Learning","date":"2025-07-17","arxiv_id":"2507.13348","repositories_listed":0,"syntology":null},{"url":null,"slug":"mgffd-vlm-multi-granularity-prompt-learning","title":"MGFFD-VLM: Multi-Granularity Prompt Learning for Face Forgery Detection with VLM","date":"2025-07-16","arxiv_id":"2507.12232","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-attribute-confusion-in-fashion","title":"Evaluating Attribute Confusion in Fashion Text-to-Image Generation","date":"2025-07-09","arxiv_id":"2507.07079","repositories_listed":0,"syntology":null},{"url":null,"slug":"linguamark-do-multimodal-models-speak-fairly","title":"LinguaMark: Do Multimodal Models Speak Fairly? A Benchmark-Based Evaluation","date":"2025-07-09","arxiv_id":"2507.07274","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-video-quality-scoring-and","title":"Bridging Video Quality Scoring and Justification via Large Multimodal Models","date":"2025-06-26","arxiv_id":"2506.21011","repositories_listed":0,"syntology":null},{"url":null,"slug":"smmile-an-expert-driven-benchmark-for","title":"SMMILE: An Expert-Driven Benchmark for Multimodal Medical In-Context Learning","date":"2025-06-26","arxiv_id":"2506.21355","repositories_listed":0,"syntology":null},{"url":"/paper/focus-internal-mllm-representations-for","slug":"focus-internal-mllm-representations-for","title":"FOCUS: Internal MLLM Representations for Efficient Fine-Grained Visual Question Answering","date":"2025-06-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"gemex-thinkvg-towards-thinking-with-visual","title":"GEMeX-ThinkVG: Towards Thinking with Visual Grounding in Medical VQA via Reinforcement Learning","date":"2025-06-22","arxiv_id":"2506.17939","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-common-vlms-rival-medical-vlms-evaluation","title":"Can Common VLMs Rival Medical VLMs? Evaluation and Strategic Insights","date":"2025-06-19","arxiv_id":"2506.17337","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-far-can-off-the-shelf-multimodal-large","title":"How Far Can Off-the-Shelf Multimodal Large Language Models Go in Online Episodic Memory Question Answering?","date":"2025-06-19","arxiv_id":"2506.16450","repositories_listed":0,"syntology":null},{"url":null,"slug":"megc2025-micro-expression-grand-challenge-on","title":"MEGC2025: Micro-Expression Grand Challenge on Spot Then Recognize and Visual Question Answering","date":"2025-06-18","arxiv_id":"2506.15298","repositories_listed":0,"syntology":null},{"url":null,"slug":"ascd-attention-steerable-contrastive-decoding","title":"ASCD: Attention-Steerable Contrastive Decoding for Reducing Hallucination in MLLM","date":"2025-06-17","arxiv_id":"2506.14766","repositories_listed":0,"syntology":null},{"url":null,"slug":"connecting-phases-of-matter-to-the-flatness","title":"Connecting phases of matter to the flatness of the loss landscape in analog variational quantum algorithms","date":"2025-06-16","arxiv_id":"2506.13865","repositories_listed":0,"syntology":null},{"url":null,"slug":"capo-reinforcing-consistent-reasoning-in","title":"CAPO: Reinforcing Consistent Reasoning in Medical Decision-Making","date":"2025-06-15","arxiv_id":"2506.12849","repositories_listed":0,"syntology":null},{"url":null,"slug":"eyesim-vqa-a-free-energy-guided-eye","title":"EyeSim-VQA: A Free-Energy-Guided Eye Simulation Framework for Video Quality Assessment","date":"2025-06-13","arxiv_id":"2506.11549","repositories_listed":0,"syntology":null},{"url":null,"slug":"halloc-token-level-localization-of-1","title":"HalLoc: Token-level Localization of Hallucinations for Vision Language Models","date":"2025-06-12","arxiv_id":"2506.10286","repositories_listed":0,"syntology":null},{"url":null,"slug":"scientists-first-exam-probing-cognitive","title":"Scientists' First Exam: Probing Cognitive Abilities of MLLM via Perception, Understanding, and Reasoning","date":"2025-06-12","arxiv_id":"2506.10521","repositories_listed":0,"syntology":null},{"url":null,"slug":"kvasir-vqa-x1-a-multimodal-dataset-for","title":"Kvasir-VQA-x1: A Multimodal Dataset for Medical Reasoning and Robust MedVQA in Gastrointestinal Endoscopy","date":"2025-06-11","arxiv_id":"2506.09958","repositories_listed":0,"syntology":null},{"url":null,"slug":"provoking-multi-modal-few-shot-lvlm-via-1","title":"Provoking Multi-modal Few-Shot LVLM via Exploration-Exploitation In-Context Learning","date":"2025-06-11","arxiv_id":"2506.09473","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-pixels-to-graphs-using-scene-and","title":"From Pixels to Graphs: using Scene and Knowledge Graphs for HD-EPIC VQA Challenge","date":"2025-06-10","arxiv_id":"2506.08553","repositories_listed":0,"syntology":null},{"url":null,"slug":"phyblock-a-progressive-benchmark-for-physical","title":"PhyBlock: A Progressive Benchmark for Physical Understanding and Planning via 3D Block Assembly","date":"2025-06-10","arxiv_id":"2506.08708","repositories_listed":0,"syntology":null},{"url":null,"slug":"lingshu-a-generalist-foundation-model-for","title":"Lingshu: A Generalist Foundation Model for Unified Multimodal Medical Understanding and Reasoning","date":"2025-06-08","arxiv_id":"2506.07044","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-04756","title":"Ontology-based knowledge representation for bone disease diagnosis: a foundation for safe and sustainable medical artificial intelligence systems","date":"2025-06-05","arxiv_id":"2506.04756","repositories_listed":0,"syntology":null},{"url":null,"slug":"rexvqa-a-large-scale-visual-question","title":"ReXVQA: A Large-scale Visual Question Answering Benchmark for Generalist Chest X-ray Understanding","date":"2025-06-04","arxiv_id":"2506.04353","repositories_listed":0,"syntology":null},{"url":null,"slug":"core-mmrag-cross-source-knowledge","title":"CoRe-MMRAG: Cross-Source Knowledge Reconciliation for Multimodal RAG","date":"2025-06-03","arxiv_id":"2506.02544","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-or-slow-integrating-fast-intuition-and","title":"Fast or Slow? Integrating Fast Intuition and Deliberate Thinking for Enhancing Visual Question Answering","date":"2025-06-01","arxiv_id":"2506.00806","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmedagent-rl-optimizing-multi-agent","title":"MMedAgent-RL: Optimizing Multi-Agent Collaboration for Multimodal Medical Reasoning","date":"2025-05-31","arxiv_id":"2506.00555","repositories_listed":0,"syntology":null},{"url":null,"slug":"proxy-fda-proxy-based-feature-distribution","title":"Proxy-FDA: Proxy-based Feature Distribution Alignment for Fine-tuning Vision Foundation Models without Forgetting","date":"2025-05-30","arxiv_id":"2505.24088","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-llms-are-bad-at-hierarchical-visual","title":"Vision LLMs Are Bad at Hierarchical Visual Understanding, and LLMs Are the Bottleneck","date":"2025-05-30","arxiv_id":"2505.24840","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-evaluation-of-multi-modal","title":"A Comprehensive Evaluation of Multi-Modal Large Language Models for Endoscopy Analysis","date":"2025-05-29","arxiv_id":"2505.23601","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmsi-bench-a-benchmark-for-multi-image","title":"MMSI-Bench: A Benchmark for Multi-Image Spatial Intelligence","date":"2025-05-29","arxiv_id":"2505.23764","repositories_listed":0,"syntology":null},{"url":null,"slug":"spoken-question-answering-for-visual-queries","title":"Spoken question answering for visual queries","date":"2025-05-29","arxiv_id":"2505.23308","repositories_listed":0,"syntology":null},{"url":null,"slug":"negvqa-can-vision-language-models-understand","title":"NegVQA: Can Vision Language Models Understand Negation?","date":"2025-05-28","arxiv_id":"2505.22946","repositories_listed":0,"syntology":null},{"url":null,"slug":"silence-is-not-consensus-disrupting-agreement","title":"Silence is Not Consensus: Disrupting Agreement Bias in Multi-Agent LLMs via Catfish Agent for Clinical Decision Making","date":"2025-05-27","arxiv_id":"2505.21503","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-multimodal-models-for","title":"Benchmarking Large Multimodal Models for Ophthalmic Visual Question Answering with OphthalWeChat","date":"2025-05-26","arxiv_id":"2505.19624","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmig-bench-towards-comprehensive-and","title":"MMIG-Bench: Towards Comprehensive and Explainable Evaluation of Multi-Modal Image Generation Models","date":"2025-05-26","arxiv_id":"2505.19415","repositories_listed":0,"syntology":null},{"url":null,"slug":"tdve-assessor-benchmarking-and-evaluating-the","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","date":"2025-05-26","arxiv_id":"2505.19535","repositories_listed":0,"syntology":null},{"url":null,"slug":"gc-kbvqa-a-new-four-stage-framework-for","title":"GC-KBVQA: A New Four-Stage Framework for Enhancing Knowledge Based Visual Question Answering Performance","date":"2025-05-25","arxiv_id":"2505.19354","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-medical-reasoning-with-curriculum","title":"Improving Medical Reasoning with Curriculum-Aware Reinforcement Learning","date":"2025-05-25","arxiv_id":"2505.19213","repositories_listed":0,"syntology":null},{"url":null,"slug":"focus-on-what-matters-enhancing-medical","title":"Focus on What Matters: Enhancing Medical Vision-Language Models with Automatic Attention Alignment Tuning","date":"2025-05-24","arxiv_id":"2505.18503","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-causal-approach-to-mitigate-modality","title":"A Causal Approach to Mitigate Modality Preference Bias in Medical Visual Question Answering","date":"2025-05-22","arxiv_id":"2505.16209","repositories_listed":0,"syntology":null},{"url":null,"slug":"ct-agent-a-multimodal-llm-agent-for-3d-ct","title":"CT-Agent: A Multimodal-LLM Agent for 3D CT Radiology Question Answering","date":"2025-05-22","arxiv_id":"2505.16229","repositories_listed":0,"syntology":null},{"url":null,"slug":"grounding-chest-x-ray-visual-question","title":"Grounding Chest X-Ray Visual Question Answering with Generated Radiology Reports","date":"2025-05-22","arxiv_id":"2505.16624","repositories_listed":0,"syntology":null},{"url":null,"slug":"medframeqa-a-multi-image-medical-vqa","title":"MedFrameQA: A Multi-Image Medical VQA Benchmark for Clinical Reasoning","date":"2025-05-22","arxiv_id":"2505.16964","repositories_listed":0,"syntology":null},{"url":null,"slug":"steering-lvlms-via-sparse-autoencoder-for","title":"Steering LVLMs via Sparse Autoencoder for Hallucination Mitigation","date":"2025-05-22","arxiv_id":"2505.16146","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-anomaly-detection-in-battery","title":"Zero-Shot Anomaly Detection in Battery Thermal Images Using Visual Question Answering with Prior Knowledge","date":"2025-05-22","arxiv_id":"2505.16674","repositories_listed":0,"syntology":null},{"url":null,"slug":"cp-llm-context-and-pixel-aware-large-language","title":"CP-LLM: Context and Pixel Aware Large Language Model for Video Quality Assessment","date":"2025-05-21","arxiv_id":"2505.16025","repositories_listed":0,"syntology":null},{"url":null,"slug":"prolonged-reasoning-is-not-all-you-need","title":"Prolonged Reasoning Is Not All You Need: Certainty-Based Adaptive Routing for Efficient LLM/MLLM Reasoning","date":"2025-05-21","arxiv_id":"2505.15154","repositories_listed":0,"syntology":null},{"url":null,"slug":"robo2vlm-visual-question-answering-from-large","title":"Robo2VLM: Visual Question Answering from Large-Scale In-the-Wild Robot Manipulation Datasets","date":"2025-05-21","arxiv_id":"2505.15517","repositories_listed":0,"syntology":null},{"url":null,"slug":"tinydrive-multiscale-visual-question","title":"TinyDrive: Multiscale Visual Question Answering with Selective Token Routing for Autonomous Driving","date":"2025-05-21","arxiv_id":"2505.15564","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-question-answering-on-multiple-remote","title":"Visual Question Answering on Multiple Remote Sensing Image Modalities","date":"2025-05-21","arxiv_id":"2505.15401","repositories_listed":0,"syntology":null},{"url":null,"slug":"debating-for-better-reasoning-an-unsupervised","title":"Debating for Better Reasoning: An Unsupervised Multimodal Approach","date":"2025-05-20","arxiv_id":"2505.14627","repositories_listed":0,"syntology":null},{"url":null,"slug":"plangpt-vl-enhancing-urban-planning-with","title":"PlanGPT-VL: Enhancing Urban Planning with Domain-Specific Vision-Language Models","date":"2025-05-20","arxiv_id":"2505.14481","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-effective-reinforcement-learning-fine","title":"Toward Effective Reinforcement Learning Fine-Tuning for Medical VQA in Vision-Language Models","date":"2025-05-20","arxiv_id":"2505.13973","repositories_listed":0,"syntology":null},{"url":null,"slug":"medsg-bench-a-benchmark-for-medical-image","title":"MedSG-Bench: A Benchmark for Medical Image Sequences Grounding","date":"2025-05-17","arxiv_id":"2505.11852","repositories_listed":0,"syntology":null},{"url":null,"slug":"tinyrs-r1-compact-multimodal-language-model","title":"TinyRS-R1: Compact Multimodal Language Model for Remote Sensing","date":"2025-05-17","arxiv_id":"2505.12099","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantically-aware-game-image-quality","title":"Semantically-Aware Game Image Quality Assessment","date":"2025-05-16","arxiv_id":"2505.11724","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-multi-image-question-answering-via","title":"Enhancing Multi-Image Question Answering via Submodular Subset Selection","date":"2025-05-15","arxiv_id":"2505.10533","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-visual-question-answering","title":"Variational Visual Question Answering","date":"2025-05-14","arxiv_id":"2505.09591","repositories_listed":0,"syntology":null},{"url":"/paper/omgm-orchestrate-multiple-granularities-and","slug":"omgm-orchestrate-multiple-granularities-and","title":"OMGM: Orchestrate Multiple Granularities and Modalities for Efficient Multimodal Retrieval","date":"2025-05-10","arxiv_id":"2505.07879","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/omgm-orchestrate-multiple-granularities-and#ran","syntology_url":"https://syntology.ai/paper/2505.07879","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07879"}},"official":null}},{"url":null,"slug":"natural-reflection-backdoor-attack-on-vision","title":"Natural Reflection Backdoor Attack on Vision Language Model for Autonomous Driving","date":"2025-05-09","arxiv_id":"2505.06413","repositories_listed":0,"syntology":null}],"record_sha256":"ed53bb835f17af050f44742f1e507c11f44e9a2f2bbc6f3d521f46c816bd259f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}