{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering-1/papers/13","list_of":"/task/visual-question-answering-1","task":"Visual Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":13,"pages_in_order":22,"rows_per_page":100,"rows":[1201,1300],"of":2177,"counts":{"archive_papers_tagged":2177,"with_a_code_link":1042,"where_syntology_ran_a_sample":378,"not_listed_spam_title":0,"listed":2177,"listed_where_code_ran":378,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":308,"every_run_a_failure_of_syntologys_instrument":70,"listed_with_a_run_with_no_instrument_failure":308,"listed_every_run_a_failure_of_syntologys_instrument":70,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering-1","prev":"/task/visual-question-answering-1/papers/12","next":"/task/visual-question-answering-1/papers/14","papers":[{"url":null,"slug":"scene-understanding-enabled-semantic","title":"Scene Understanding Enabled Semantic Communication with Open Channel Coding","date":"2025-01-24","arxiv_id":"2501.14520","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-knowledge-graph-and-llms-for","title":"Combining Knowledge Graph and LLMs for Enhanced Zero-shot Visual Question Answering","date":"2025-01-22","arxiv_id":"2501.12697","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-multimodal-llms-do-visual-temporal","title":"Can Multimodal LLMs do Visual Temporal Understanding and Reasoning? The answer is No!","date":"2025-01-18","arxiv_id":"2501.10674","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-visual-defense-adversarial-pre","title":"Double Visual Defense: Adversarial Pre-training and Instruction Tuning for Improving Vision-Language Model Robustness","date":"2025-01-16","arxiv_id":"2501.09446","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-knowledge-integration-for-enhanced","title":"Dynamic Knowledge Integration for Enhanced Vision-Language Reasoning","date":"2025-01-15","arxiv_id":"2501.08597","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-scene-understanding-for-vision","title":"Embodied Scene Understanding for Vision Language Models via MetaVQA","date":"2025-01-15","arxiv_id":"2501.09167","repositories_listed":0,"syntology":null},{"url":null,"slug":"sar-strikes-back-a-new-hope-for-rsvqa","title":"SAR Strikes Back: A New Hope for RSVQA","date":"2025-01-14","arxiv_id":"2501.08131","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-quest-for-visual-understanding-a-journey","title":"The Quest for Visual Understanding: A Journey Through the Evolution of Visual Question Answering","date":"2025-01-13","arxiv_id":"2501.07109","repositories_listed":0,"syntology":null},{"url":null,"slug":"geopix-multi-modal-large-language-model-for","title":"GeoPix: Multi-Modal Large Language Model for Pixel-level Image Understanding in Remote Sensing","date":"2025-01-12","arxiv_id":"2501.06828","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-language-priors-for-visual","title":"Overcoming Language Priors for Visual Question Answering Based on Knowledge Distillation","date":"2025-01-10","arxiv_id":"2501.05690","repositories_listed":0,"syntology":null},{"url":null,"slug":"llava-octopus-unlocking-instruction-driven","title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","date":"2025-01-09","arxiv_id":"2501.05067","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervision-free-vision-language-alignment","title":"Feedback-Driven Vision-Language Alignment with Minimal Human Supervision","date":"2025-01-08","arxiv_id":"2501.04568","repositories_listed":0,"syntology":null},{"url":"/paper/kanoclip-zero-shot-anomaly-detection-through","slug":"kanoclip-zero-shot-anomaly-detection-through","title":"KAnoCLIP: Zero-Shot Anomaly Detection through Knowledge-Driven Prompt Learning and Enhanced Cross-Modal Integration","date":"2025-01-07","arxiv_id":"2501.03786","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-question-answering-from-early","title":"Visual question answering: from early developments to recent advances -- a survey","date":"2025-01-07","arxiv_id":"2501.03939","repositories_listed":0,"syntology":null},{"url":null,"slug":"accounting-for-focus-ambiguity-in-visual","title":"Accounting for Focus Ambiguity in Visual Questions","date":"2025-01-04","arxiv_id":"2501.02201","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-face-anti-spoofing-enhancing","title":"Interpretable Face Anti-Spoofing: Enhancing Generalization with Multimodal Large Language Models","date":"2025-01-03","arxiv_id":"2501.01720","repositories_listed":0,"syntology":null},{"url":null,"slug":"mocoll-agent-based-specific-and-general-model","title":"MoColl: Agent-Based Specific and General Model Collaboration for Image Captioning","date":"2025-01-03","arxiv_id":"2501.01834","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-up-clip-based-unanswerable-problem","title":"CLIP-UP: CLIP-Based Unanswerable Problem Detection for Visual Question Answering","date":"2025-01-02","arxiv_id":"2501.01371","repositories_listed":0,"syntology":null},{"url":null,"slug":"adadare-gamma-balancing-stability-and","title":"AdaDARE-gamma: Balancing Stability and Plasticity in Multi-modal LLMs through Efficient Adaptation","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"alignment-mining-and-fusion-representation","title":"Alignment, Mining and Fusion: Representation Alignment with Hard Negative Mining and Selective Knowledge Fusion for Medical Visual Question Answering","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"efficientllava-generalizable-auto-pruning-for-1","title":"EfficientLLaVA: Generalizable Auto-Pruning for Large Vision-language Models","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"jtd-uav-mllm-enhanced-joint-tracking-and","title":"JTD-UAV: MLLM-Enhanced Joint Tracking and Description Framework for Anti-UAV Systems","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"separation-of-powers-on-segregating-knowledge","title":"Separation of Powers: On Segregating Knowledge from Observation in LLM-enabled Knowledge-based Visual Question Answering","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-visual-language-priors-in-vlms","title":"Probing Visual Language Priors in VLMs","date":"2024-12-31","arxiv_id":"2501.00569","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-multimodal-rag-llm-for-accurate","title":"Enhanced Multimodal RAG-LLM for Accurate Visual Question Answering","date":"2024-12-30","arxiv_id":"2412.20927","repositories_listed":0,"syntology":null},{"url":null,"slug":"ergochat-a-visual-query-system-for-the","title":"ErgoChat: a Visual Query System for the Ergonomic Risk Assessment of Construction Workers","date":"2024-12-27","arxiv_id":"2412.19954","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agents-based-on-large-language-models","title":"Multi-Agents Based on Large Language Models for Knowledge-based Visual Question Answering","date":"2024-12-24","arxiv_id":"2412.18351","repositories_listed":0,"syntology":null},{"url":null,"slug":"textmatch-enhancing-image-text-consistency","title":"TextMatch: Enhancing Image-Text Consistency Through Multimodal Optimization","date":"2024-12-24","arxiv_id":"2412.18185","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-text-rich-visual-comprehension","title":"Cross-Lingual Text-Rich Visual Comprehension: An Information Theory Perspective","date":"2024-12-23","arxiv_id":"2412.17787","repositories_listed":0,"syntology":null},{"url":null,"slug":"ffa-sora-video-generation-as-fundus","title":"FFA Sora, video generation as fundus fluorescein angiography simulator","date":"2024-12-23","arxiv_id":"2412.17346","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-of-large-multimodal-model-datasets","title":"Survey of Large Multimodal Model Datasets, Application Categories and Taxonomy","date":"2024-12-23","arxiv_id":"2412.17759","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-large-language-models-with-1","title":"Prompting Large Language Models with Rationale Heuristics for Knowledge-based Visual Question Answering","date":"2024-12-22","arxiv_id":"2412.16936","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedpia-permuting-and-integrating-adapters","title":"FedPIA -- Permuting and Integrating Adapters leveraging Wasserstein Barycenters for Finetuning Foundation Models in Multi-Modal Federated Learning","date":"2024-12-19","arxiv_id":"2412.14424","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-concept-centric-approach-to-multi-modality","title":"A Concept-Centric Approach to Multi-Modality Learning","date":"2024-12-18","arxiv_id":"2412.13847","repositories_listed":0,"syntology":null},{"url":null,"slug":"cpath-omni-a-unified-multimodal-foundation","title":"CPath-Omni: A Unified Multimodal Foundation Model for Patch and Whole Slide Image Analysis in Computational Pathology","date":"2024-12-16","arxiv_id":"2412.12077","repositories_listed":0,"syntology":null},{"url":null,"slug":"overview-of-trec-2024-medical-video-question","title":"Overview of TREC 2024 Medical Video Question Answering (MedVidQA) Track","date":"2024-12-15","arxiv_id":"2412.11056","repositories_listed":0,"syntology":null},{"url":null,"slug":"damage-assessment-after-natural-disasters","title":"Damage Assessment after Natural Disasters with UAVs: Semantic Feature Extraction using Deep Learning","date":"2024-12-14","arxiv_id":"2412.10756","repositories_listed":0,"syntology":null},{"url":null,"slug":"patch-level-sounding-object-tracking-for","title":"Patch-level Sounding Object Tracking for Audio-Visual Question Answering","date":"2024-12-14","arxiv_id":"2412.10749","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlr-bench-multilingual-benchmark-dataset-for","title":"VLR-Bench: Multilingual Benchmark Dataset for Vision-Language Retrieval Augmented Generation","date":"2024-12-13","arxiv_id":"2412.10151","repositories_listed":0,"syntology":null},{"url":null,"slug":"viunit-visual-unit-tests-for-more-robust","title":"ViUniT: Visual Unit Tests for More Robust Visual Programming","date":"2024-12-12","arxiv_id":"2412.08859","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multimodal-social-agent","title":"A Multimodal Social Agent","date":"2024-12-11","arxiv_id":"2501.06189","repositories_listed":0,"syntology":null},{"url":null,"slug":"barking-up-the-syntactic-tree-enhancing-vlm","title":"Barking Up The Syntactic Tree: Enhancing VLM Training with Syntactic Losses","date":"2024-12-11","arxiv_id":"2412.08110","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-generate-visual-programs-without","title":"Can We Generate Visual Programs Without Prompting LLMs?","date":"2024-12-11","arxiv_id":"2412.08564","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-vision-language-tasks-benefit-from-large","title":"How Vision-Language Tasks Benefit from Large Pre-trained Models: A Survey","date":"2024-12-11","arxiv_id":"2412.08158","repositories_listed":0,"syntology":null},{"url":"/paper/illume-illuminating-your-llms-to-see-draw-and","slug":"illume-illuminating-your-llms-to-see-draw-and","title":"ILLUME: Illuminating Your LLMs to See, Draw, and Self-Enhance","date":"2024-12-09","arxiv_id":"2412.06673","repositories_listed":0,"syntology":null},{"url":null,"slug":"ranked-from-within-ranking-large-multimodal","title":"Ranked from Within: Ranking Large Multimodal Models for Visual Question Answering Without Labels","date":"2024-12-09","arxiv_id":"2412.06461","repositories_listed":0,"syntology":null},{"url":null,"slug":"eaco-enhancing-alignment-in-multimodal-llms","title":"EACO: Enhancing Alignment in Multimodal LLMs via Critical Observation","date":"2024-12-06","arxiv_id":"2412.04903","repositories_listed":0,"syntology":null},{"url":null,"slug":"findings-of-the-second-babylm-challenge","title":"Findings of the Second BabyLM Challenge: Sample-Efficient Pretraining on Developmentally Plausible Corpora","date":"2024-12-06","arxiv_id":"2412.05149","repositories_listed":0,"syntology":null},{"url":null,"slug":"t2i-factualbench-benchmarking-the-factuality","title":"T2I-FactualBench: Benchmarking the Factuality of Text-to-Image Models with Knowledge-Intensive Concepts","date":"2024-12-05","arxiv_id":"2412.04300","repositories_listed":0,"syntology":null},{"url":null,"slug":"cegi-measuring-the-trade-off-between","title":"CEGI: Measuring the trade-off between efficiency and carbon emissions for SLMs and VLMs","date":"2024-12-03","arxiv_id":"2412.02602","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-and-interpretable-multimodal","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","date":"2024-12-03","arxiv_id":"2412.02104","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-world-s-museums-through","title":"Understanding the World's Museums through Vision-Language Reasoning","date":"2024-12-02","arxiv_id":"2412.01370","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-logit-lens-contextual-embeddings-for","title":"Beyond Logit Lens: Contextual Embeddings for Robust Hallucination Detection & Grounding in VLMs","date":"2024-11-28","arxiv_id":"2411.19187","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-attention-vectors-generative","title":"Sparse Attention Vectors: Generative Multimodal Model Features Are Discriminative Vision-Language Classifiers","date":"2024-11-28","arxiv_id":"2412.00142","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-data-curation-effectively-distills","title":"Active Data Curation Effectively Distills Large-Scale Multimodal Models","date":"2024-11-27","arxiv_id":"2411.18674","repositories_listed":0,"syntology":null},{"url":null,"slug":"electrovizqa-how-well-do-multi-modal-llms","title":"ElectroVizQA: How well do Multi-modal LLMs perform in Electronics Visual Question Answering?","date":"2024-11-27","arxiv_id":"2412.00102","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-multi-modal-large-language-models","title":"Efficient Multi-modal Large Language Models via Visual Token Grouping","date":"2024-11-26","arxiv_id":"2411.17773","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-language-understanding-and-inference","title":"Natural Language Understanding and Inference with MLLM in Visual Question Answering: A Survey","date":"2024-11-26","arxiv_id":"2411.17558","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-progressive-curriculum-learning-for","title":"Task Progressive Curriculum Learning for Robust Visual Question Answering","date":"2024-11-26","arxiv_id":"2411.17292","repositories_listed":0,"syntology":null},{"url":null,"slug":"gemex-a-large-scale-groundable-and","title":"GEMeX: A Large-Scale, Groundable, and Explainable Medical VQA Benchmark for Chest X-ray Diagnosis","date":"2024-11-25","arxiv_id":"2411.16778","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-guided-coarse-to-fine-fusion-network-for","title":"Text-Guided Coarse-to-Fine Fusion Network for Robust Remote Sensing Visual Question Answering","date":"2024-11-24","arxiv_id":"2411.15770","repositories_listed":0,"syntology":null},{"url":null,"slug":"finecaption-compositional-image-captioning","title":"FINECAPTION: Compositional Image Captioning Focusing on Wherever You Want at Any Granularity","date":"2024-11-23","arxiv_id":"2411.15411","repositories_listed":0,"syntology":null},{"url":null,"slug":"freepruner-a-training-free-approach-for-large","title":"freePruner: A Training-free Approach for Large Multimodal Model Acceleration","date":"2024-11-23","arxiv_id":"2411.15446","repositories_listed":0,"syntology":null},{"url":null,"slug":"rewind-understanding-long-videos-with","title":"ReWind: Understanding Long Videos with Instructed Learnable Memory","date":"2024-11-23","arxiv_id":"2411.15556","repositories_listed":0,"syntology":null},{"url":"/paper/focusllava-a-coarse-to-fine-approach-for","slug":"focusllava-a-coarse-to-fine-approach-for","title":"FocusLLaVA: A Coarse-to-Fine Approach for Efficient and Effective Visual Token Compression","date":"2024-11-21","arxiv_id":"2411.14228","repositories_listed":0,"syntology":null},{"url":"/paper/looking-beyond-text-reducing-language-bias-in","slug":"looking-beyond-text-reducing-language-bias-in","title":"Looking Beyond Text: Reducing Language bias in Large Vision-Language Models via Multimodal Dual-Attention and Soft-Image Guidance","date":"2024-11-21","arxiv_id":"2411.14279","repositories_listed":0,"syntology":null},{"url":null,"slug":"lavida-drive-vision-text-interaction-vlm-for","title":"LaVida Drive: Vision-Text Interaction VLM for Autonomous Driving with Token Selection, Recovery and Enhancement","date":"2024-11-20","arxiv_id":"2411.12980","repositories_listed":0,"syntology":null},{"url":null,"slug":"uni-mlip-unified-self-supervision-for-medical","title":"Uni-Mlip: Unified Self-supervision for Medical Vision Language Pre-training","date":"2024-11-20","arxiv_id":"2411.15207","repositories_listed":0,"syntology":null},{"url":null,"slug":"catch-complementary-adaptive-token-level","title":"CATCH: Complementary Adaptive Token-level Contrastive Decoding to Mitigate Hallucinations in LVLMs","date":"2024-11-19","arxiv_id":"2411.12713","repositories_listed":0,"syntology":null},{"url":null,"slug":"med-2e3-a-2d-enhanced-3d-medical-multimodal","title":"Med-2E3: A 2D-Enhanced 3D Medical Multimodal Large Language Model","date":"2024-11-19","arxiv_id":"2411.12783","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-survey-on-visual-question","title":"A Comprehensive Survey on Visual Question Answering Datasets and Algorithms","date":"2024-11-17","arxiv_id":"2411.11150","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-augmented-multimodal-llms-for-surgical","title":"Memory-Augmented Multimodal LLMs for Surgical VQA via Self-Contained Inquiry","date":"2024-11-17","arxiv_id":"2411.10937","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-multimodal-llms-the-mechanistic","title":"Understanding Multimodal LLMs: the Mechanistic Interpretability of Llava in Visual Question Answering","date":"2024-11-17","arxiv_id":"2411.10950","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-vision-language-models-for-remote","title":"Large Vision-Language Models for Remote Sensing Visual Question Answering","date":"2024-11-16","arxiv_id":"2411.10857","repositories_listed":0,"syntology":null},{"url":null,"slug":"amxfp4-taming-activation-outliers-with","title":"AMXFP4: Taming Activation Outliers with Asymmetric Microscaling Floating-Point for 4-bit LLM Inference","date":"2024-11-15","arxiv_id":"2411.09909","repositories_listed":0,"syntology":null},{"url":null,"slug":"everything-is-a-video-unifying-modalities","title":"Everything is a Video: Unifying Modalities through Next-Frame Prediction","date":"2024-11-15","arxiv_id":"2411.10503","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-question-answering-based-evaluation","title":"Visual question answering based evaluation metrics for text-to-image generation","date":"2024-11-15","arxiv_id":"2411.10183","repositories_listed":0,"syntology":null},{"url":"/paper/aligned-vector-quantization-for-edge-cloud","slug":"aligned-vector-quantization-for-edge-cloud","title":"Aligned Vector Quantization for Edge-Cloud Collabrative Vision-Language Models","date":"2024-11-08","arxiv_id":"2411.05961","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-object-detection-modality-into","title":"Integrating Object Detection Modality into Visual Language Model for Enhanced Autonomous Driving Agent","date":"2024-11-08","arxiv_id":"2411.05898","repositories_listed":0,"syntology":null},{"url":null,"slug":"m3docrag-multi-modal-retrieval-is-what-you","title":"M3DocRAG: Multi-modal Retrieval is What You Need for Multi-page Multi-document Understanding","date":"2024-11-07","arxiv_id":"2411.04952","repositories_listed":0,"syntology":null},{"url":null,"slug":"sasr-net-source-aware-semantic-representation","title":"SaSR-Net: Source-Aware Semantic Representation Network for Enhancing Audio-Visual Question Answering","date":"2024-11-07","arxiv_id":"2411.04933","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-is-deceiving-exploitation-of-visual","title":"Seeing is Deceiving: Exploitation of Visual Pathways in Multi-Modal Language Models","date":"2024-11-07","arxiv_id":"2411.05056","repositories_listed":0,"syntology":null},{"url":null,"slug":"neurips-2023-competition-privacy-preserving","title":"NeurIPS 2023 Competition: Privacy Preserving Federated Learning Document VQA","date":"2024-11-06","arxiv_id":"2411.03730","repositories_listed":0,"syntology":null},{"url":null,"slug":"select2plan-training-free-icl-based-planning","title":"Select2Plan: Training-Free ICL-Based Planning through VQA and Memory Retrieval","date":"2024-11-06","arxiv_id":"2411.04006","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-pixels-to-prose-advancing-multi-modal","title":"From Pixels to Prose: Advancing Multi-Modal Language Models for Remote Sensing","date":"2024-11-05","arxiv_id":"2411.05826","repositories_listed":0,"syntology":null},{"url":null,"slug":"mme-finance-a-multimodal-finance-benchmark","title":"MME-Finance: A Multimodal Finance Benchmark for Expert-level Understanding and Reasoning","date":"2024-11-05","arxiv_id":"2411.03314","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-commonsense-knowledge-distillation","title":"Multimodal Commonsense Knowledge Distillation for Visual Question Answering","date":"2024-11-05","arxiv_id":"2411.02722","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-vlm-to-keep-it-learning-generation-and","title":"One VLM to Keep it Learning: Generation and Balancing for Data-free Continual Visual Question Answering","date":"2024-11-04","arxiv_id":"2411.02210","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-visual-question-answering-method-for-sar","title":"A Visual Question Answering Method for SAR Ship: Breaking the Requirement for Multimodal Dataset Construction and Model Fine-Tuning","date":"2024-11-03","arxiv_id":"2411.01445","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-oriented-semantic-communication-for-1","title":"Goal-Oriented Semantic Communication for Wireless Visual Question Answering","date":"2024-11-03","arxiv_id":"2411.02452","repositories_listed":0,"syntology":null},{"url":null,"slug":"rs-moe-mixture-of-experts-for-remote-sensing","title":"RS-MoE: Mixture of Experts for Remote Sensing Image Captioning and Visual Question Answering","date":"2024-11-03","arxiv_id":"2411.01595","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-a-robust-radiology-report","title":"Designing a Robust Radiology Report Generation System","date":"2024-11-02","arxiv_id":"2411.01153","repositories_listed":0,"syntology":null},{"url":null,"slug":"simpsonsvqa-enhancing-inquiry-based-learning","title":"SimpsonsVQA: Enhancing Inquiry-Based Learning with a Tailored Dataset","date":"2024-10-30","arxiv_id":"2410.22648","repositories_listed":0,"syntology":null},{"url":null,"slug":"grade-quantifying-sample-diversity-in-text-to","title":"GRADE: Quantifying Sample Diversity in Text-to-Image Models","date":"2024-10-29","arxiv_id":"2410.22592","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-overlap-is-responsible-for-the","title":"Attention Overlap Is Responsible for The Entity Missing Problem in Text-to-image Diffusion Models!","date":"2024-10-28","arxiv_id":"2410.20972","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-bilinear-attention-based-fusion-for","title":"Efficient Bilinear Attention-based Fusion for Medical Visual Question Answering","date":"2024-10-28","arxiv_id":"2410.21000","repositories_listed":0,"syntology":null},{"url":null,"slug":"face-mllm-a-large-face-perception-model","title":"Face-MLLM: A Large Face Perception Model","date":"2024-10-28","arxiv_id":"2410.20717","repositories_listed":0,"syntology":null},{"url":null,"slug":"r-llava-improving-med-vqa-understanding","title":"R-LLaVA: Improving Med-VQA Understanding through Visual Region of Interest","date":"2024-10-27","arxiv_id":"2410.20327","repositories_listed":0,"syntology":null},{"url":null,"slug":"give-guiding-visual-encoder-to-perceive","title":"GiVE: Guiding Visual Encoder to Perceive Overlooked Information","date":"2024-10-26","arxiv_id":"2410.20109","repositories_listed":0,"syntology":null},{"url":null,"slug":"sensor2text-enabling-natural-language","title":"Sensor2Text: Enabling Natural Language Interactions for Daily Activity Tracking Using Wearable Sensors","date":"2024-10-26","arxiv_id":"2410.20034","repositories_listed":0,"syntology":null}],"record_sha256":"2d0961ccfcadcb4ed90ef14959d50de4c4fb29944b3e203118304716a4fa82e0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}