{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering/papers/12","list_of":"/task/visual-question-answering","task":"Visual Question Answering (VQA)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":12,"pages_in_order":22,"rows_per_page":100,"rows":[1101,1200],"of":2167,"counts":{"archive_papers_tagged":2167,"with_a_code_link":1039,"where_syntology_ran_a_sample":359,"not_listed_spam_title":0,"listed":2167,"listed_where_code_ran":359,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":287,"every_run_a_failure_of_syntologys_instrument":72,"listed_with_a_run_with_no_instrument_failure":287,"listed_every_run_a_failure_of_syntologys_instrument":72,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering","prev":"/task/visual-question-answering/papers/11","next":"/task/visual-question-answering/papers/13","papers":[{"url":null,"slug":"r-3-vqa-read-the-room-by-video-social","title":"R^3-VQA: \"Read the Room\" by Video Social Reasoning","date":"2025-05-07","arxiv_id":"2505.04147","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffvqa-video-quality-assessment-using","title":"DiffVQA: Video Quality Assessment Using Diffusion Feature Extractor","date":"2025-05-06","arxiv_id":"2505.03261","repositories_listed":0,"syntology":null},{"url":null,"slug":"aor-anatomical-ontology-guided-reasoning-for","title":"AOR: Anatomical Ontology-Guided Reasoning for Medical Large Multimodal Model in Chest X-Ray Interpretation","date":"2025-05-05","arxiv_id":"2505.02830","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-semantic-communication-in-large","title":"Task-Oriented Semantic Communication in Large Multimodal Models-based Vehicle Networks","date":"2025-05-05","arxiv_id":"2505.02413","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatialllm-a-compound-3d-informed-design","title":"SpatialLLM: A Compound 3D-Informed Design towards Spatially-Intelligent Large Multimodal Models","date":"2025-05-01","arxiv_id":"2505.00788","repositories_listed":0,"syntology":null},{"url":null,"slug":"localizing-before-answering-a-hallucination","title":"Localizing Before Answering: A Hallucination Evaluation Benchmark for Grounded Medical Multimodal LLMs","date":"2025-04-30","arxiv_id":"2505.00744","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-survey-of-knowledge-based","title":"A Comprehensive Survey of Knowledge-Based Vision Question Answering Systems: The Lifecycle of Knowledge in Visual Reasoning Task","date":"2025-04-24","arxiv_id":"2504.17547","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-calibration-of-prediction-sets-in","title":"Data-Driven Calibration of Prediction Sets in Large Vision-Language Models Based on Inductive Conformal Prediction","date":"2025-04-24","arxiv_id":"2504.17671","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-semantic-gaps-improving-medical","title":"Bridging the Semantic Gaps: Improving Medical VQA Consistency with LLM-Augmented Question Sets","date":"2025-04-16","arxiv_id":"2504.11777","repositories_listed":0,"syntology":null},{"url":null,"slug":"dvlta-vqa-decoupled-vision-language-modeling","title":"DVLTA-VQA: Decoupled Vision-Language Modeling with Text-Guided Adaptation for Blind Video Quality Assessment","date":"2025-04-16","arxiv_id":"2504.11733","repositories_listed":0,"syntology":null},{"url":null,"slug":"instruction-augmented-multimodal-alignment","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","date":"2025-04-16","arxiv_id":"2504.12018","repositories_listed":0,"syntology":null},{"url":null,"slug":"puzzlebench-a-fully-dynamic-evaluation","title":"PuzzleBench: A Fully Dynamic Evaluation Framework for Large Multimodal Models on Puzzle Solving","date":"2025-04-15","arxiv_id":"2504.10885","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-trustworthy-multimodal-ai-a-review","title":"Building Trustworthy Multimodal AI: A Review of Fairness, Transparency, and Ethics in Vision-Language Tasks","date":"2025-04-14","arxiv_id":"2504.13199","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmkb-rag-a-multi-modal-knowledge-based","title":"MMKB-RAG: A Multi-Modal Knowledge-Based Retrieval-Augmented Generation Framework","date":"2025-04-14","arxiv_id":"2504.10074","repositories_listed":0,"syntology":null},{"url":null,"slug":"notes-bank-benchmarking-neural-transcription","title":"NoTeS-Bank: Benchmarking Neural Transcription and Search for Scientific Notes Understanding","date":"2025-04-12","arxiv_id":"2504.09249","repositories_listed":0,"syntology":null},{"url":null,"slug":"pathvlm-r1-a-reinforcement-learning-driven","title":"PathVLM-R1: A Reinforcement Learning-Driven Reasoning Model for Pathology Visual-Language Tasks","date":"2025-04-12","arxiv_id":"2504.09258","repositories_listed":0,"syntology":null},{"url":null,"slug":"tokenfocus-vqa-enhancing-text-to-image","title":"TokenFocus-VQA: Enhancing Text-to-Image Alignment with Position-Aware Focus and Multi-Perspective Aggregations on LVLMs","date":"2025-04-10","arxiv_id":"2504.07556","repositories_listed":0,"syntology":null},{"url":null,"slug":"unirvqa-a-unified-framework-for-retrieval","title":"UniRVQA: A Unified Framework for Retrieval-Augmented Vision Question Answering via Self-Reflective Joint Training","date":"2025-04-05","arxiv_id":"2504.04065","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-modeling-for-medical-visual","title":"Hierarchical Modeling for Medical Visual Question Answering with Cross-Attention Fusion","date":"2025-04-04","arxiv_id":"2504.03135","repositories_listed":0,"syntology":null},{"url":null,"slug":"qirl-boosting-visual-question-answering-via","title":"QIRL: Boosting Visual Question Answering via Optimized Question-Image Relation Learning","date":"2025-04-04","arxiv_id":"2504.03337","repositories_listed":0,"syntology":null},{"url":null,"slug":"socialgesture-delving-into-multi-person","title":"SocialGesture: Delving into Multi-person Gesture Understanding","date":"2025-04-03","arxiv_id":"2504.02244","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-llms-for-user-aware-multimodal","title":"Reasoning LLMs for User-Aware Multimodal Conversational Agents","date":"2025-04-02","arxiv_id":"2504.01700","repositories_listed":0,"syntology":null},{"url":null,"slug":"mpdrive-improving-spatial-understanding-with","title":"MPDrive: Improving Spatial Understanding with Marker-Based Prompt Learning for Autonomous Driving","date":"2025-04-01","arxiv_id":"2504.00379","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-well-can-vison-language-models-understand","title":"How Well Can Vison-Language Models Understand Humans' Intention? An Open-ended Theory of Mind Question Evaluation Benchmark","date":"2025-03-28","arxiv_id":"2503.22093","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature4x-bridging-any-monocular-video-to-4d","title":"Feature4X: Bridging Any Monocular Video to 4D Agentic AI with Versatile Gaussian Feature Fields","date":"2025-03-26","arxiv_id":"2503.20776","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-amplified-semantic-entropy-for","title":"Vision-Amplified Semantic Entropy for Hallucination Detection in Medical Visual Question Answering","date":"2025-03-26","arxiv_id":"2503.20504","repositories_listed":0,"syntology":null},{"url":null,"slug":"lego-puzzles-how-good-are-mllms-at-multi-step","title":"LEGO-Puzzles: How Good Are MLLMs at Multi-Step Spatial Reasoning?","date":"2025-03-25","arxiv_id":"2503.19990","repositories_listed":0,"syntology":null},{"url":"/paper/orion-a-holistic-end-to-end-autonomous","slug":"orion-a-holistic-end-to-end-autonomous","title":"ORION: A Holistic End-to-End Autonomous Driving Framework by Vision-Language Instructed Action Generation","date":"2025-03-25","arxiv_id":"2503.19755","repositories_listed":0,"syntology":null},{"url":null,"slug":"din-diffusion-model-for-robust-medical-vqa","title":"DiN: Diffusion Model for Robust Medical VQA with Semantic Noisy Labels","date":"2025-03-24","arxiv_id":"2503.18536","repositories_listed":0,"syntology":null},{"url":null,"slug":"magic-vqa-multimodal-and-grounded-inference","title":"MAGIC-VQA: Multimodal And Grounded Inference with Commonsense Knowledge for Visual Question Answering","date":"2025-03-24","arxiv_id":"2503.18491","repositories_listed":0,"syntology":null},{"url":null,"slug":"where-is-this-coming-from-making-groundedness","title":"Where is this coming from? Making groundedness count in the evaluation of Document VQA models","date":"2025-03-24","arxiv_id":"2503.19120","repositories_listed":0,"syntology":null},{"url":null,"slug":"expanding-the-boundaries-of-vision-prior","title":"Expanding the Boundaries of Vision Prior Knowledge in Multi-modal Large Language Models","date":"2025-03-23","arxiv_id":"2503.18034","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-vision-centric-remote-sensing-benchmark","title":"A Vision Centric Remote Sensing Benchmark","date":"2025-03-20","arxiv_id":"2503.15816","repositories_listed":0,"syntology":null},{"url":null,"slug":"truthlens-a-training-free-paradigm-for","title":"TruthLens:A Training-Free Paradigm for DeepFake Detection","date":"2025-03-19","arxiv_id":"2503.15342","repositories_listed":0,"syntology":null},{"url":null,"slug":"upme-an-unsupervised-peer-review-framework","title":"UPME: An Unsupervised Peer Review Framework for Multimodal Large Language Model Evaluation","date":"2025-03-19","arxiv_id":"2503.14941","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatbev-a-visual-language-model-that","title":"ChatBEV: A Visual Language Model that Understands BEV Maps","date":"2025-03-18","arxiv_id":"2503.13938","repositories_listed":0,"syntology":null},{"url":"/paper/marten-visual-question-answering-with-mask","slug":"marten-visual-question-answering-with-mask","title":"Marten: Visual Question Answering with Mask Generation for Multi-modal Document Understanding","date":"2025-03-18","arxiv_id":"2503.14140","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/marten-visual-question-answering-with-mask#ran","syntology_url":"https://syntology.ai/paper/2503.14140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.14140"}},"official":null}},{"url":null,"slug":"georsmllm-a-multimodal-large-language-model","title":"GeoRSMLLM: A Multimodal Large Language Model for Vision-Language Tasks in Geoscience and Remote Sensing","date":"2025-03-16","arxiv_id":"2503.12490","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynrsl-vlm-enhancing-autonomous-driving","title":"DynRsl-VLM: Enhancing Autonomous Driving Perception with Dynamic Resolution Vision-Language Models","date":"2025-03-14","arxiv_id":"2503.11265","repositories_listed":0,"syntology":null},{"url":null,"slug":"astrea-a-moe-based-visual-understanding-model","title":"Astrea: A MOE-based Visual Understanding Model with Progressive Alignment","date":"2025-03-12","arxiv_id":"2503.09445","repositories_listed":0,"syntology":null},{"url":null,"slug":"surgicalvlm-agent-towards-an-interactive-ai","title":"SurgicalVLM-Agent: Towards an Interactive AI Co-Pilot for Pituitary Surgery","date":"2025-03-12","arxiv_id":"2503.09474","repositories_listed":0,"syntology":null},{"url":null,"slug":"bring-remote-sensing-object-detect-into","title":"Bring Remote Sensing Object Detect Into Nature Language Model: Using SFT Method","date":"2025-03-11","arxiv_id":"2503.08144","repositories_listed":0,"syntology":null},{"url":null,"slug":"comicspap-understanding-comic-strips-by","title":"ComicsPAP: understanding comic strips by picking the correct panel","date":"2025-03-11","arxiv_id":"2503.08561","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-and-reasoning-with-confidence","title":"Seeing and Reasoning with Confidence: Supercharging Multimodal LLMs with an Uncertainty-Aware Agentic Framework","date":"2025-03-11","arxiv_id":"2503.08308","repositories_listed":0,"syntology":null},{"url":null,"slug":"robusto-1-dataset-comparing-humans-and-vlms","title":"Robusto-1 Dataset: Comparing Humans and VLMs on real out-of-distribution Autonomous Driving VQA from Peru","date":"2025-03-10","arxiv_id":"2503.07587","repositories_listed":0,"syntology":null},{"url":null,"slug":"callireader-contextualizing-chinese","title":"CalliReader: Contextualizing Chinese Calligraphy via an Embedding-Aligned Vision-Language Model","date":"2025-03-09","arxiv_id":"2503.06472","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-frequency-domain-representations","title":"Integrating Frequency-Domain Representations with Low-Rank Adaptation in Vision-Language Models","date":"2025-03-08","arxiv_id":"2503.06003","repositories_listed":0,"syntology":null},{"url":null,"slug":"moemoe-question-guided-dense-and-scalable","title":"MoEMoE: Question Guided Dense and Scalable Sparse Mixture-of-Expert for Multi-source Multi-modal Answering","date":"2025-03-08","arxiv_id":"2503.06296","repositories_listed":0,"syntology":null},{"url":null,"slug":"splattalk-3d-vqa-with-gaussian-splatting","title":"SplatTalk: 3D VQA with Gaussian Splatting","date":"2025-03-08","arxiv_id":"2503.06271","repositories_listed":0,"syntology":null},{"url":"/paper/a-token-level-text-image-foundation-model-for","slug":"a-token-level-text-image-foundation-model-for","title":"A Token-level Text Image Foundation Model for Document Understanding","date":"2025-03-04","arxiv_id":"2503.02304","repositories_listed":0,"syntology":{"n":6,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-token-level-text-image-foundation-model-for#ran","syntology_url":"https://syntology.ai/paper/2503.02304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.02304"}},"official":null}},{"url":null,"slug":"enhancing-multi-hop-reasoning-in-vision","title":"Enhancing Multi-hop Reasoning in Vision-Language Models via Self-Distillation with Multi-Prompt Ensembling","date":"2025-03-03","arxiv_id":"2503.01754","repositories_listed":0,"syntology":null},{"url":null,"slug":"v-2-dial-unification-of-video-and-visual","title":"V$^2$Dial: Unification of Video and Visual Dialog via Multimodal Experts","date":"2025-03-03","arxiv_id":"2503.02063","repositories_listed":0,"syntology":null},{"url":null,"slug":"funbench-benchmarking-fundus-reading-skills","title":"FunBench: Benchmarking Fundus Reading Skills of MLLMs","date":"2025-03-02","arxiv_id":"2503.00901","repositories_listed":0,"syntology":null},{"url":null,"slug":"abc-achieving-better-control-of-multimodal","title":"ABC: Achieving Better Control of Multimodal Embeddings using VLMs","date":"2025-03-01","arxiv_id":"2503.00329","repositories_listed":0,"syntology":null},{"url":null,"slug":"cl-moe-enhancing-multimodal-large-language","title":"CL-MoE: Enhancing Multimodal Large Language Model with Dual Momentum Mixture-of-Experts for Continual Visual Question Answering","date":"2025-03-01","arxiv_id":"2503.00413","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-retrieval-augmented-generation","title":"Fine-Grained Retrieval-Augmented Generation for Visual Question Answering","date":"2025-02-28","arxiv_id":"2502.20964","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatreid-open-ended-interactive-person","title":"ChatReID: Open-ended Interactive Person Retrieval via Hierarchical Progressive Tuning for Vision Language Models","date":"2025-02-27","arxiv_id":"2502.19958","repositories_listed":0,"syntology":null},{"url":null,"slug":"talking-to-the-brain-using-large-language","title":"Talking to the brain: Using Large Language Models as Proxies to Model Brain Semantic Representation","date":"2025-02-26","arxiv_id":"2502.18725","repositories_listed":0,"syntology":null},{"url":null,"slug":"filterrag-zero-shot-informed-retrieval","title":"FilterRAG: Zero-Shot Informed Retrieval-Augmented Generation to Mitigate Hallucinations in VQA","date":"2025-02-25","arxiv_id":"2502.18536","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-augmented-visual-question-answering-1","title":"Retrieval-Augmented Visual Question Answering via Built-in Autoregressive Search Engines","date":"2025-02-23","arxiv_id":"2502.16641","repositories_listed":0,"syntology":null},{"url":null,"slug":"directional-gradient-projection-for-robust","title":"Directional Gradient Projection for Robust Fine-Tuning of Foundation Models","date":"2025-02-21","arxiv_id":"2502.15895","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-advanced-techniques-for-visual","title":"Exploring Advanced Techniques for Visual Question Answering: A Comprehensive Comparison","date":"2025-02-20","arxiv_id":"2502.14827","repositories_listed":0,"syntology":null},{"url":null,"slug":"hardware-friendly-static-quantization-method","title":"Hardware-Friendly Static Quantization Method for Video Diffusion Transformers","date":"2025-02-20","arxiv_id":"2502.15077","repositories_listed":0,"syntology":null},{"url":null,"slug":"sce2drivex-a-generalized-mllm-framework-for","title":"Sce2DriveX: A Generalized MLLM Framework for Scene-to-Drive Learning","date":"2025-02-19","arxiv_id":"2502.14917","repositories_listed":0,"syntology":null},{"url":null,"slug":"safeeraser-enhancing-safety-in-multimodal","title":"SafeEraser: Enhancing Safety in Multimodal Large Language Models through Multimodal Machine Unlearning","date":"2025-02-18","arxiv_id":"2502.12520","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-retrieval-augmentation-for-open","title":"Multi-Modal Retrieval Augmentation for Open-Ended and Knowledge-Intensive Video Question Answering","date":"2025-02-17","arxiv_id":"2502.11747","repositories_listed":0,"syntology":null},{"url":"/paper/viscon-100k-leveraging-contextual-web-data","slug":"viscon-100k-leveraging-contextual-web-data","title":"VisCon-100K: Leveraging Contextual Web Data for Fine-tuning Vision Language Models","date":"2025-02-14","arxiv_id":"2502.10250","repositories_listed":0,"syntology":null},{"url":null,"slug":"abduction-of-domain-relationships-from-data","title":"Abduction of Domain Relationships from Data for VQA","date":"2025-02-13","arxiv_id":"2502.09219","repositories_listed":0,"syntology":null},{"url":null,"slug":"emoassist-emotional-assistant-for-visual","title":"EmoAssist: Emotional Assistant for Visual Impairment Community","date":"2025-02-13","arxiv_id":"2502.09285","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-graph-question-answering-with-asp-and","title":"Visual Graph Question Answering with ASP and LLMs for Language Parsing","date":"2025-02-13","arxiv_id":"2502.09211","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-analysis-of-traditional-vqa","title":"Performance Analysis of Traditional VQA Models Under Limited Computational Resources","date":"2025-02-09","arxiv_id":"2502.05738","repositories_listed":0,"syntology":null},{"url":null,"slug":"hummingbird-high-fidelity-image-generation","title":"Hummingbird: High Fidelity Image Generation via Multimodal Context Alignment","date":"2025-02-07","arxiv_id":"2502.05153","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-few-shot-continual-learning-in","title":"Efficient Few-Shot Continual Learning in Vision-Language Models","date":"2025-02-06","arxiv_id":"2502.04098","repositories_listed":0,"syntology":null},{"url":null,"slug":"hd-epic-a-highly-detailed-egocentric-video","title":"HD-EPIC: A Highly-Detailed Egocentric Video Dataset","date":"2025-02-06","arxiv_id":"2502.04144","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypo3d-exploring-hypothetical-reasoning-in-3d","title":"Hypo3D: Exploring Hypothetical Reasoning in 3D","date":"2025-02-02","arxiv_id":"2502.00954","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlm-assisted-continual-learning-for-visual","title":"VLM-Assisted Continual learning for Visual Question Answering in Self-Driving","date":"2025-02-02","arxiv_id":"2502.00843","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-large-vision-language-models-for","title":"Scaling Large Vision-Language Models for Enhanced Multimodal Comprehension In Biomedical Image Analysis","date":"2025-01-26","arxiv_id":"2501.15370","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-understanding-enabled-semantic","title":"Scene Understanding Enabled Semantic Communication with Open Channel Coding","date":"2025-01-24","arxiv_id":"2501.14520","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-knowledge-graph-and-llms-for","title":"Combining Knowledge Graph and LLMs for Enhanced Zero-shot Visual Question Answering","date":"2025-01-22","arxiv_id":"2501.12697","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-multimodal-llms-do-visual-temporal","title":"Can Multimodal LLMs do Visual Temporal Understanding and Reasoning? The answer is No!","date":"2025-01-18","arxiv_id":"2501.10674","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-scene-understanding-for-vision","title":"Embodied Scene Understanding for Vision Language Models via MetaVQA","date":"2025-01-15","arxiv_id":"2501.09167","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-quest-for-visual-understanding-a-journey","title":"The Quest for Visual Understanding: A Journey Through the Evolution of Visual Question Answering","date":"2025-01-13","arxiv_id":"2501.07109","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-quality-assessment-for-online","title":"Video Quality Assessment for Online Processing: From Spatial to Temporal Sampling","date":"2025-01-13","arxiv_id":"2501.07087","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-language-priors-for-visual","title":"Overcoming Language Priors for Visual Question Answering Based on Knowledge Distillation","date":"2025-01-10","arxiv_id":"2501.05690","repositories_listed":0,"syntology":null},{"url":null,"slug":"commonsense-video-question-answering-through","title":"Commonsense Video Question Answering through Video-Grounded Entailment Tree Reasoning","date":"2025-01-09","arxiv_id":"2501.05069","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-financial-vqa-in-vision-language","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","date":"2025-01-08","arxiv_id":"2501.04675","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-question-answering-from-early","title":"Visual question answering: from early developments to recent advances -- a survey","date":"2025-01-07","arxiv_id":"2501.03939","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilevel-semantic-aware-model-for-ai","title":"Multilevel Semantic-Aware Model for AI-Generated Video Quality Assessment","date":"2025-01-06","arxiv_id":"2501.02706","repositories_listed":0,"syntology":null},{"url":null,"slug":"accounting-for-focus-ambiguity-in-visual","title":"Accounting for Focus Ambiguity in Visual Questions","date":"2025-01-04","arxiv_id":"2501.02201","repositories_listed":0,"syntology":null},{"url":null,"slug":"guiding-medical-vision-language-models-with","title":"Guiding Medical Vision-Language Models with Explicit Visual Prompts: Framework Design and Comprehensive Exploration of Prompt Variations","date":"2025-01-04","arxiv_id":"2501.02385","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-face-anti-spoofing-enhancing","title":"Interpretable Face Anti-Spoofing: Enhancing Generalization with Multimodal Large Language Models","date":"2025-01-03","arxiv_id":"2501.01720","repositories_listed":0,"syntology":null},{"url":null,"slug":"mocoll-agent-based-specific-and-general-model","title":"MoColl: Agent-Based Specific and General Model Collaboration for Image Captioning","date":"2025-01-03","arxiv_id":"2501.01834","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-up-clip-based-unanswerable-problem","title":"CLIP-UP: CLIP-Based Unanswerable Problem Detection for Visual Question Answering","date":"2025-01-02","arxiv_id":"2501.01371","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignment-mining-and-fusion-representation","title":"Alignment, Mining and Fusion: Representation Alignment with Hard Negative Mining and Selective Knowledge Fusion for Medical Visual Question Answering","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"f-3ocus-federated-finetuning-of-vision","title":"F^3OCUS - Federated Finetuning of Vision-Language Foundation Models with Optimal Client Layer Updating Strategy via Multi-objective Meta-Heuristics","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"jtd-uav-mllm-enhanced-joint-tracking-and","title":"JTD-UAV: MLLM-Enhanced Joint Tracking and Description Framework for Anti-UAV Systems","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"v-2dial-unification-of-video-and-visual","title":"V^2Dial: Unification of Video and Visual Dialog via Multimodal Experts","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-visual-language-priors-in-vlms","title":"Probing Visual Language Priors in VLMs","date":"2024-12-31","arxiv_id":"2501.00569","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-multimodal-rag-llm-for-accurate","title":"Enhanced Multimodal RAG-LLM for Accurate Visual Question Answering","date":"2024-12-30","arxiv_id":"2412.20927","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-layer-selective-transfer","title":"Investigating layer-selective transfer learning of QAOA parameters for Max-Cut problem","date":"2024-12-30","arxiv_id":"2412.21071","repositories_listed":0,"syntology":null}],"record_sha256":"67c46bb94ea3b08bd2278c0524e7486ef83ce2a35f2c8fba1aaa058115a1114d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}