{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering-1/papers/11","list_of":"/task/visual-question-answering-1","task":"Visual Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":22,"rows_per_page":100,"rows":[1001,1100],"of":2177,"counts":{"archive_papers_tagged":2177,"with_a_code_link":1042,"where_syntology_ran_a_sample":378,"not_listed_spam_title":0,"listed":2177,"listed_where_code_ran":378,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":308,"every_run_a_failure_of_syntologys_instrument":70,"listed_with_a_run_with_no_instrument_failure":308,"listed_every_run_a_failure_of_syntologys_instrument":70,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering-1","prev":"/task/visual-question-answering-1/papers/10","next":"/task/visual-question-answering-1/papers/12","papers":[{"url":"/paper/latent-alignment-and-variational-attention","slug":"latent-alignment-and-variational-attention","title":"Latent Alignment and Variational Attention","date":"2018-07-10","arxiv_id":"1807.03756","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/latent-alignment-and-variational-attention#ran","syntology_url":"https://syntology.ai/paper/1807.03756","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.03756"}},"official":{"repos":["harvardnlp/var-attn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/nmt-keras-a-very-flexible-toolkit-with-a","slug":"nmt-keras-a-very-flexible-toolkit-with-a","title":"NMT-Keras: a Very Flexible Toolkit with a Focus on Interactive NMT and Online Learning","date":"2018-07-09","arxiv_id":"1807.03096","repositories_listed":1,"syntology":null},{"url":"/paper/semantically-equivalent-adversarial-rules-for","slug":"semantically-equivalent-adversarial-rules-for","title":"Semantically Equivalent Adversarial Rules for Debugging NLP models","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-conditioned-graph-structures-for","slug":"learning-conditioned-graph-structures-for","title":"Learning Conditioned Graph Structures for Interpretable Visual Question Answering","date":"2018-06-19","arxiv_id":"1806.07243","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-conditioned-graph-structures-for#ran","syntology_url":"https://syntology.ai/paper/1806.07243","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.07243"}},"official":{"repos":["aimbrain/vqa-project"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/iparaphrasing-extracting-visually-grounded","slug":"iparaphrasing-extracting-visually-grounded","title":"iParaphrasing: Extracting Visually Grounded Paraphrases via an Image","date":"2018-06-12","arxiv_id":"1806.04284","repositories_listed":1,"syntology":null},{"url":"/paper/r-vqa-learning-visual-relation-facts-with","slug":"r-vqa-learning-visual-relation-facts-with","title":"R-VQA: Learning Visual Relation Facts with Semantic Attention for Visual Question Answering","date":"2018-05-24","arxiv_id":"1805.09701","repositories_listed":1,"syntology":null},{"url":"/paper/improved-fusion-of-visual-and-language","slug":"improved-fusion-of-visual-and-language","title":"Improved Fusion of Visual and Language Representations by Dense Symmetric Co-Attention for Visual Question Answering","date":"2018-04-03","arxiv_id":"1804.00775","repositories_listed":1,"syntology":null},{"url":"/paper/differential-attention-for-visual-question","slug":"differential-attention-for-visual-question","title":"Differential Attention for Visual Question Answering","date":"2018-04-01","arxiv_id":"1804.00298","repositories_listed":1,"syntology":null},{"url":"/paper/transparency-by-design-closing-the-gap","slug":"transparency-by-design-closing-the-gap","title":"Transparency by Design: Closing the Gap Between Performance and Interpretability in Visual Reasoning","date":"2018-03-14","arxiv_id":"1803.05268","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/transparency-by-design-closing-the-gap#ran","syntology_url":"https://syntology.ai/paper/1803.05268","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.05268"}},"official":{"repos":["davidmascharka/tbd-nets"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vizwiz-grand-challenge-answering-visual","slug":"vizwiz-grand-challenge-answering-visual","title":"VizWiz Grand Challenge: Answering Visual Questions from Blind People","date":"2018-02-22","arxiv_id":"1802.08218","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-count-objects-in-natural-images","slug":"learning-to-count-objects-in-natural-images","title":"Learning to Count Objects in Natural Images for Visual Question Answering","date":"2018-02-15","arxiv_id":"1802.05766","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-explanations-justifying-decisions","slug":"multimodal-explanations-justifying-decisions","title":"Multimodal Explanations: Justifying Decisions and Pointing to the Evidence","date":"2018-02-15","arxiv_id":"1802.08129","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-explanations-justifying-decisions#ran","syntology_url":"https://syntology.ai/paper/1802.08129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.08129"}},"official":{"repos":["Seth-Park/MultimodalExplanations"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-recurrent-attention-units-for-visual","slug":"dual-recurrent-attention-units-for-visual","title":"Dual Recurrent Attention Units for Visual Question Answering","date":"2018-02-01","arxiv_id":"1802.00209","repositories_listed":1,"syntology":null},{"url":"/paper/game-of-sketches-deep-recurrent-models-of","slug":"game-of-sketches-deep-recurrent-models-of","title":"Game of Sketches: Deep Recurrent Models of Pictionary-style Word Guessing","date":"2018-01-29","arxiv_id":"1801.09356","repositories_listed":1,"syntology":null},{"url":"/paper/dvqa-understanding-data-visualizations-via","slug":"dvqa-understanding-data-visualizations-via","title":"DVQA: Understanding Data Visualizations via Question Answering","date":"2018-01-24","arxiv_id":"1801.08163","repositories_listed":1,"syntology":null},{"url":"/paper/structured-triplet-learning-with-pos-tag","slug":"structured-triplet-learning-with-pos-tag","title":"Structured Triplet Learning with POS-tag Guided Attention for Visual Question Answering","date":"2018-01-24","arxiv_id":"1801.07853","repositories_listed":1,"syntology":null},{"url":"/paper/iqa-visual-question-answering-in-interactive","slug":"iqa-visual-question-answering-in-interactive","title":"IQA: Visual Question Answering in Interactive Environments","date":"2017-12-09","arxiv_id":"1712.03316","repositories_listed":1,"syntology":null},{"url":"/paper/dont-just-assume-look-and-answer-overcoming","slug":"dont-just-assume-look-and-answer-overcoming","title":"Don't Just Assume; Look and Answer: Overcoming Priors for Visual Question Answering","date":"2017-12-01","arxiv_id":"1712.00377","repositories_listed":1,"syntology":null},{"url":"/paper/locally-smoothed-neural-networks","slug":"locally-smoothed-neural-networks","title":"Locally Smoothed Neural Networks","date":"2017-11-22","arxiv_id":"1711.08132","repositories_listed":1,"syntology":null},{"url":"/paper/co-attending-free-form-regions-and-detections","slug":"co-attending-free-form-regions-and-detections","title":"Co-attending Free-form Regions and Detections with Multi-modal Multiplicative Feature Embedding for Visual Question Answering","date":"2017-11-18","arxiv_id":"1711.06794","repositories_listed":1,"syntology":null},{"url":"/paper/co-attending-regions-and-detections-with","slug":"co-attending-regions-and-detections-with","title":"Co-attending Regions and Detections with Multi-modal Multiplicative Embedding for VQA","date":"2017-11-18","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/high-order-attention-models-for-visual","slug":"high-order-attention-models-for-visual","title":"High-Order Attention Models for Visual Question Answering","date":"2017-11-12","arxiv_id":"1711.04323","repositories_listed":1,"syntology":null},{"url":"/paper/active-learning-for-visual-question-answering","slug":"active-learning-for-visual-question-answering","title":"Active Learning for Visual Question Answering: An Empirical Study","date":"2017-11-06","arxiv_id":"1711.01732","repositories_listed":1,"syntology":null},{"url":"/paper/vqs-linking-segmentations-to-questions-and","slug":"vqs-linking-segmentations-to-questions-and","title":"VQS: Linking Segmentations to Questions and Answers for Supervised Attention in VQA and Question-Focused Semantic Segmentation","date":"2017-08-15","arxiv_id":"1708.04686","repositories_listed":1,"syntology":null},{"url":"/paper/structured-attentions-for-visual-question","slug":"structured-attentions-for-visual-question","title":"Structured Attentions for Visual Question Answering","date":"2017-08-07","arxiv_id":"1708.02071","repositories_listed":1,"syntology":null},{"url":"/paper/effective-approaches-to-batch-parallelization","slug":"effective-approaches-to-batch-parallelization","title":"Effective Approaches to Batch Parallelization for Dynamic Neural Network Architectures","date":"2017-07-08","arxiv_id":"1707.02402","repositories_listed":1,"syntology":null},{"url":"/paper/learning-convolutional-text-representations","slug":"learning-convolutional-text-representations","title":"Learning Convolutional Text Representations for Visual Question Answering","date":"2017-05-18","arxiv_id":"1705.06824","repositories_listed":1,"syntology":null},{"url":"/paper/speech-based-visual-question-answering","slug":"speech-based-visual-question-answering","title":"Speech-Based Visual Question Answering","date":"2017-05-01","arxiv_id":"1705.00464","repositories_listed":1,"syntology":null},{"url":"/paper/the-promise-of-premise-harnessing-question","slug":"the-promise-of-premise-harnessing-question","title":"The Promise of Premise: Harnessing Question Premises in Visual Question Answering","date":"2017-05-01","arxiv_id":"1705.00601","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-reason-end-to-end-module-networks","slug":"learning-to-reason-end-to-end-module-networks","title":"Learning to Reason: End-to-End Module Networks for Visual Question Answering","date":"2017-04-18","arxiv_id":"1704.05526","repositories_listed":1,"syntology":null},{"url":"/paper/whats-in-a-question-using-visual-questions-as","slug":"whats-in-a-question-using-visual-questions-as","title":"What's in a Question: Using Visual Questions as a Form of Supervision","date":"2017-04-12","arxiv_id":"1704.03895","repositories_listed":1,"syntology":null},{"url":"/paper/vibiknet-visual-bidirectional-kernelized","slug":"vibiknet-visual-bidirectional-kernelized","title":"VIBIKNet: Visual Bidirectional Kernelized Network for Visual Question Answering","date":"2016-12-12","arxiv_id":"1612.03628","repositories_listed":1,"syntology":null},{"url":"/paper/open-ended-visual-question-answering","slug":"open-ended-visual-question-answering","title":"Open-Ended Visual Question-Answering","date":"2016-10-09","arxiv_id":"1610.02692","repositories_listed":1,"syntology":null},{"url":"/paper/visual-question-answering-datasets-algorithms","slug":"visual-question-answering-datasets-algorithms","title":"Visual Question Answering: Datasets, Algorithms, and Future Challenges","date":"2016-10-05","arxiv_id":"1610.01465","repositories_listed":1,"syntology":null},{"url":"/paper/visual-question-answering-a-survey-of-methods","slug":"visual-question-answering-a-survey-of-methods","title":"Visual Question Answering: A Survey of Methods and Datasets","date":"2016-07-20","arxiv_id":"1607.05910","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-the-behavior-of-visual-question","slug":"analyzing-the-behavior-of-visual-question","title":"Analyzing the Behavior of Visual Question Answering Models","date":"2016-06-23","arxiv_id":"1606.07356","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/analyzing-the-behavior-of-visual-question#ran","syntology_url":"https://syntology.ai/paper/1606.07356","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.07356"}},"official":{"repos":["akirafukui/vqa-mcb"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-residual-learning-for-visual-qa","slug":"multimodal-residual-learning-for-visual-qa","title":"Multimodal Residual Learning for Visual QA","date":"2016-06-05","arxiv_id":"1606.01455","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-instance-segmentation-with","slug":"end-to-end-instance-segmentation-with","title":"End-to-End Instance Segmentation with Recurrent Attention","date":"2016-05-30","arxiv_id":"1605.09410","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/end-to-end-instance-segmentation-with#ran","syntology_url":"https://syntology.ai/paper/1605.09410","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1605.09410"}},"official":null}},{"url":"/paper/ask-your-neurons-a-deep-learning-approach-to","slug":"ask-your-neurons-a-deep-learning-approach-to","title":"Ask Your Neurons: A Deep Learning Approach to Visual Question Answering","date":"2016-05-09","arxiv_id":"1605.02697","repositories_listed":1,"syntology":null},{"url":"/paper/counting-everyday-objects-in-everyday-scenes","slug":"counting-everyday-objects-in-everyday-scenes","title":"Counting Everyday Objects in Everyday Scenes","date":"2016-04-12","arxiv_id":"1604.03505","repositories_listed":1,"syntology":null},{"url":"/paper/ask-attend-and-answer-exploring-question","slug":"ask-attend-and-answer-exploring-question","title":"Ask, Attend and Answer: Exploring Question-Guided Spatial Attention for Visual Question Answering","date":"2015-11-17","arxiv_id":"1511.05234","repositories_listed":1,"syntology":null},{"url":"/paper/what-value-do-explicit-high-level-concepts","slug":"what-value-do-explicit-high-level-concepts","title":"What value do explicit high level concepts have in vision to language problems?","date":"2015-06-03","arxiv_id":"1506.01144","repositories_listed":1,"syntology":null},{"url":null,"slug":"barriers-in-integrating-medical-visual","title":"Barriers in Integrating Medical Visual Question Answering into Radiology Workflows: A Scoping Review and Clinicians' Insights","date":"2025-07-09","arxiv_id":"2507.08036","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-attribute-confusion-in-fashion","title":"Evaluating Attribute Confusion in Fashion Text-to-Image Generation","date":"2025-07-09","arxiv_id":"2507.07079","repositories_listed":0,"syntology":null},{"url":null,"slug":"linguamark-do-multimodal-models-speak-fairly","title":"LinguaMark: Do Multimodal Models Speak Fairly? A Benchmark-Based Evaluation","date":"2025-07-09","arxiv_id":"2507.07274","repositories_listed":0,"syntology":null},{"url":null,"slug":"magic-evaluating-multimodal-cognition-toward","title":"MagiC: Evaluating Multimodal Cognition Toward Grounded Visual Reasoning","date":"2025-07-09","arxiv_id":"2507.07297","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-scientific-visual-question","title":"Enhancing Scientific Visual Question Answering through Multimodal Reasoning and Ensemble Modeling","date":"2025-07-08","arxiv_id":"2507.06183","repositories_listed":0,"syntology":null},{"url":null,"slug":"reloop-seeing-twice-and-thinking-backwards","title":"ReLoop: \"Seeing Twice and Thinking Backwards\" via Closed-loop Training to Mitigate Hallucinations in Multimodal understanding","date":"2025-07-07","arxiv_id":"2507.04943","repositories_listed":0,"syntology":null},{"url":null,"slug":"smmile-an-expert-driven-benchmark-for","title":"SMMILE: An Expert-Driven Benchmark for Multimodal Medical In-Context Learning","date":"2025-06-26","arxiv_id":"2506.21355","repositories_listed":0,"syntology":null},{"url":"/paper/focus-internal-mllm-representations-for","slug":"focus-internal-mllm-representations-for","title":"FOCUS: Internal MLLM Representations for Efficient Fine-Grained Visual Question Answering","date":"2025-06-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-enhanced-modality-asymmetric","title":"Semantic-enhanced Modality-asymmetric Retrieval for Online E-commerce Search","date":"2025-06-25","arxiv_id":"2506.20330","repositories_listed":0,"syntology":null},{"url":null,"slug":"gemex-thinkvg-towards-thinking-with-visual","title":"GEMeX-ThinkVG: Towards Thinking with Visual Grounding in Medical VQA via Reinforcement Learning","date":"2025-06-22","arxiv_id":"2506.17939","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-r1-video-grounded-large-language-models","title":"Scene-R1: Video-Grounded Large Language Models for 3D Scene Reasoning without 3D Annotations","date":"2025-06-21","arxiv_id":"2506.17545","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-common-vlms-rival-medical-vlms-evaluation","title":"Can Common VLMs Rival Medical VLMs? Evaluation and Strategic Insights","date":"2025-06-19","arxiv_id":"2506.17337","repositories_listed":0,"syntology":null},{"url":null,"slug":"megc2025-micro-expression-grand-challenge-on","title":"MEGC2025: Micro-Expression Grand Challenge on Spot Then Recognize and Visual Question Answering","date":"2025-06-18","arxiv_id":"2506.15298","repositories_listed":0,"syntology":null},{"url":null,"slug":"capo-reinforcing-consistent-reasoning-in","title":"CAPO: Reinforcing Consistent Reasoning in Medical Decision-Making","date":"2025-06-15","arxiv_id":"2506.12849","repositories_listed":0,"syntology":null},{"url":null,"slug":"antigrounding-lifting-robotic-actions-into","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","date":"2025-06-14","arxiv_id":"2506.12374","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-fast-reliable-and-secure-programming","title":"A Fast, Reliable, and Secure Programming Language for LLM Agents with Code Actions","date":"2025-06-13","arxiv_id":"2506.12202","repositories_listed":0,"syntology":null},{"url":null,"slug":"mtabvqa-evaluating-multi-tabular-reasoning-of","title":"MTabVQA: Evaluating Multi-Tabular Reasoning of Language Models in Visual Space","date":"2025-06-13","arxiv_id":"2506.11684","repositories_listed":0,"syntology":null},{"url":null,"slug":"halloc-token-level-localization-of-1","title":"HalLoc: Token-level Localization of Hallucinations for Vision Language Models","date":"2025-06-12","arxiv_id":"2506.10286","repositories_listed":0,"syntology":null},{"url":null,"slug":"kvasir-vqa-x1-a-multimodal-dataset-for","title":"Kvasir-VQA-x1: A Multimodal Dataset for Medical Reasoning and Robust MedVQA in Gastrointestinal Endoscopy","date":"2025-06-11","arxiv_id":"2506.09958","repositories_listed":0,"syntology":null},{"url":null,"slug":"provoking-multi-modal-few-shot-lvlm-via-1","title":"Provoking Multi-modal Few-Shot LVLM via Exploration-Exploitation In-Context Learning","date":"2025-06-11","arxiv_id":"2506.09473","repositories_listed":0,"syntology":null},{"url":"/paper/an-open-source-software-toolkit-benchmark","slug":"an-open-source-software-toolkit-benchmark","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","date":"2025-06-10","arxiv_id":"2506.09172","repositories_listed":0,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/an-open-source-software-toolkit-benchmark#ran","syntology_url":"https://syntology.ai/paper/2506.09172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09172"}},"official":null}},{"url":null,"slug":"phyblock-a-progressive-benchmark-for-physical","title":"PhyBlock: A Progressive Benchmark for Physical Understanding and Planning via 3D Block Assembly","date":"2025-06-10","arxiv_id":"2506.08708","repositories_listed":0,"syntology":null},{"url":null,"slug":"hallucination-at-a-glance-controlled-visual","title":"Hallucination at a Glance: Controlled Visual Edits and Fine-Grained Multimodal Learning","date":"2025-06-08","arxiv_id":"2506.07227","repositories_listed":0,"syntology":null},{"url":null,"slug":"lingshu-a-generalist-foundation-model-for","title":"Lingshu: A Generalist Foundation Model for Unified Multimodal Medical Understanding and Reasoning","date":"2025-06-08","arxiv_id":"2506.07044","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-adaptive-prompt-distillation-for-few","title":"Meta-Adaptive Prompt Distillation for Few-Shot Visual Question Answering","date":"2025-06-07","arxiv_id":"2506.06905","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-04756","title":"Ontology-based knowledge representation for bone disease diagnosis: a foundation for safe and sustainable medical artificial intelligence systems","date":"2025-06-05","arxiv_id":"2506.04756","repositories_listed":0,"syntology":null},{"url":null,"slug":"textvidbench-a-benchmark-for-long-video-scene","title":"TextVidBench: A Benchmark for Long Video Scene Text Understanding","date":"2025-06-05","arxiv_id":"2506.04983","repositories_listed":0,"syntology":null},{"url":null,"slug":"rexvqa-a-large-scale-visual-question","title":"ReXVQA: A Large-scale Visual Question Answering Benchmark for Generalist Chest X-ray Understanding","date":"2025-06-04","arxiv_id":"2506.04353","repositories_listed":0,"syntology":null},{"url":null,"slug":"hanfu-bench-a-multimodal-benchmark-on-cross","title":"Hanfu-Bench: A Multimodal Benchmark on Cross-Temporal Cultural Understanding and Transcreation","date":"2025-06-02","arxiv_id":"2506.01565","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-sparsity-for-effective-and-efficient","title":"Learning Sparsity for Effective and Efficient Music Performance Question Answering","date":"2025-06-02","arxiv_id":"2506.01319","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-or-slow-integrating-fast-intuition-and","title":"Fast or Slow? Integrating Fast Intuition and Deliberate Thinking for Enhancing Visual Question Answering","date":"2025-06-01","arxiv_id":"2506.00806","repositories_listed":0,"syntology":null},{"url":null,"slug":"light-as-deception-gpt-driven-natural","title":"Light as Deception: GPT-driven Natural Relighting Against Vision-Language Pre-training Models","date":"2025-05-30","arxiv_id":"2505.24227","repositories_listed":0,"syntology":null},{"url":null,"slug":"medorch-medical-diagnosis-with-tool-augmented","title":"MedOrch: Medical Diagnosis with Tool-Augmented Reasoning Agents for Flexible Extensibility","date":"2025-05-30","arxiv_id":"2506.00235","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-llms-are-bad-at-hierarchical-visual","title":"Vision LLMs Are Bad at Hierarchical Visual Understanding, and LLMs Are the Bottleneck","date":"2025-05-30","arxiv_id":"2505.24840","repositories_listed":0,"syntology":null},{"url":null,"slug":"mrag-elucidating-the-design-space-of-multi","title":"mRAG: Elucidating the Design Space of Multi-modal Retrieval-Augmented Generation","date":"2025-05-29","arxiv_id":"2505.24073","repositories_listed":0,"syntology":null},{"url":null,"slug":"negvqa-can-vision-language-models-understand","title":"NegVQA: Can Vision Language Models Understand Negation?","date":"2025-05-28","arxiv_id":"2505.22946","repositories_listed":0,"syntology":null},{"url":null,"slug":"music-s-multimodal-complexity-in-avqa-why-we","title":"Music's Multimodal Complexity in AVQA: Why We Need More than General Multimodal LLMs","date":"2025-05-27","arxiv_id":"2505.20638","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-multimodal-models-for","title":"Benchmarking Large Multimodal Models for Ophthalmic Visual Question Answering with OphthalWeChat","date":"2025-05-26","arxiv_id":"2505.19624","repositories_listed":0,"syntology":null},{"url":null,"slug":"gc-kbvqa-a-new-four-stage-framework-for","title":"GC-KBVQA: A New Four-Stage Framework for Enhancing Knowledge Based Visual Question Answering Performance","date":"2025-05-25","arxiv_id":"2505.19354","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-causal-approach-to-mitigate-modality","title":"A Causal Approach to Mitigate Modality Preference Bias in Medical Visual Question Answering","date":"2025-05-22","arxiv_id":"2505.16209","repositories_listed":0,"syntology":null},{"url":null,"slug":"ct-agent-a-multimodal-llm-agent-for-3d-ct","title":"CT-Agent: A Multimodal-LLM Agent for 3D CT Radiology Question Answering","date":"2025-05-22","arxiv_id":"2505.16229","repositories_listed":0,"syntology":null},{"url":null,"slug":"grounding-chest-x-ray-visual-question","title":"Grounding Chest X-Ray Visual Question Answering with Generated Radiology Reports","date":"2025-05-22","arxiv_id":"2505.16624","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-far-and-clearly-mitigating","title":"Seeing Far and Clearly: Mitigating Hallucinations in MLLMs with Attention Causal Decoding","date":"2025-05-22","arxiv_id":"2505.16652","repositories_listed":0,"syntology":null},{"url":null,"slug":"steering-lvlms-via-sparse-autoencoder-for","title":"Steering LVLMs via Sparse Autoencoder for Hallucination Mitigation","date":"2025-05-22","arxiv_id":"2505.16146","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-anomaly-detection-in-battery","title":"Zero-Shot Anomaly Detection in Battery Thermal Images Using Visual Question Answering with Prior Knowledge","date":"2025-05-22","arxiv_id":"2505.16674","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-pathology-rationale-and-token","title":"Discovering Pathology Rationale and Token Allocation for Efficient Multimodal Pathology Reasoning","date":"2025-05-21","arxiv_id":"2505.15687","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-centered-interactive-learning-via-mllms-1","title":"Human-centered Interactive Learning via MLLMs for Text-to-Image Person Re-identification","date":"2025-05-21","arxiv_id":"2506.11036","repositories_listed":0,"syntology":null},{"url":null,"slug":"robo2vlm-visual-question-answering-from-large","title":"Robo2VLM: Visual Question Answering from Large-Scale In-the-Wild Robot Manipulation Datasets","date":"2025-05-21","arxiv_id":"2505.15517","repositories_listed":0,"syntology":null},{"url":null,"slug":"tinydrive-multiscale-visual-question","title":"TinyDrive: Multiscale Visual Question Answering with Selective Token Routing for Autonomous Driving","date":"2025-05-21","arxiv_id":"2505.15564","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-question-answering-on-multiple-remote","title":"Visual Question Answering on Multiple Remote Sensing Image Modalities","date":"2025-05-21","arxiv_id":"2505.15401","repositories_listed":0,"syntology":null},{"url":null,"slug":"debating-for-better-reasoning-an-unsupervised","title":"Debating for Better Reasoning: An Unsupervised Multimodal Approach","date":"2025-05-20","arxiv_id":"2505.14627","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adaptation-of-vlm-for-soccer-video","title":"Domain Adaptation of VLM for Soccer Video Understanding","date":"2025-05-20","arxiv_id":"2505.13860","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-effective-reinforcement-learning-fine","title":"Toward Effective Reinforcement Learning Fine-Tuning for Medical VQA in Vision-Language Models","date":"2025-05-20","arxiv_id":"2505.13973","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-omnidirectional-reasoning-with-360-r1","title":"Towards Omnidirectional Reasoning with 360-R1: A Dataset, Benchmark, and GRPO-based Method","date":"2025-05-20","arxiv_id":"2505.14197","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-complexity-in-videoqa-via","title":"Understanding Complexity in VideoQA via Visual Program Generation","date":"2025-05-19","arxiv_id":"2505.13429","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-vision-tokenizer-tuning","title":"End-to-End Vision Tokenizer Tuning","date":"2025-05-15","arxiv_id":"2505.10562","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-visual-question-answering","title":"Variational Visual Question Answering","date":"2025-05-14","arxiv_id":"2505.09591","repositories_listed":0,"syntology":null},{"url":null,"slug":"visually-interpretable-subtask-reasoning-for","title":"Visually Interpretable Subtask Reasoning for Visual Question Answering","date":"2025-05-12","arxiv_id":"2505.08084","repositories_listed":0,"syntology":null}],"record_sha256":"96cd09e5445e8547b9934799b4f4ae9feedfe2e50d945aeedb1e54db3ca904dc","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}