{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multimodal-large-language-model/papers/3","list_of":"/task/multimodal-large-language-model","task":"Multimodal Large Language Model","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":4,"rows_per_page":100,"rows":[201,300],"of":347,"counts":{"archive_papers_tagged":347,"with_a_code_link":160,"where_syntology_ran_a_sample":59,"not_listed_spam_title":0,"listed":347,"listed_where_code_ran":59,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":51,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":51,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multimodal-large-language-model","prev":"/task/multimodal-large-language-model/papers/2","next":"/task/multimodal-large-language-model/papers/4","papers":[{"url":null,"slug":"cleanmap-distilling-multimodal-llms-for","title":"CleanMAP: Distilling Multimodal LLMs for Confidence-Driven Crowdsourced HD Map Updates","date":"2025-04-14","arxiv_id":"2504.10738","repositories_listed":0,"syntology":null},{"url":null,"slug":"mavors-multi-granularity-video-representation","title":"Mavors: Multi-granularity Video Representation for Multimodal Large Language Model","date":"2025-04-14","arxiv_id":"2504.10068","repositories_listed":0,"syntology":null},{"url":null,"slug":"marmot-multi-agent-reasoning-for-multi-object","title":"Marmot: Multi-Agent Reasoning for Multi-Object Self-Correcting in Improving Image-Text Alignment","date":"2025-04-10","arxiv_id":"2504.20054","repositories_listed":0,"syntology":null},{"url":null,"slug":"face-llava-facial-expression-and-attribute","title":"Face-LLaVA: Facial Expression and Attribute Understanding through Instruction Tuning","date":"2025-04-09","arxiv_id":"2504.07198","repositories_listed":0,"syntology":null},{"url":null,"slug":"movsam-a-single-image-moving-object","title":"MovSAM: A Single-image Moving Object Segmentation Framework Based on Deep Thinking","date":"2025-04-09","arxiv_id":"2504.06863","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-agent-quality-driven-chain-of-thought-image","title":"Q-Agent: Quality-Driven Chain-of-Thought Image Restoration Agent through Robust Multimodal Large Language Model","date":"2025-04-09","arxiv_id":"2504.07148","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-visual-text-grounding-of-multimodal","title":"Towards Visual Text Grounding of Multimodal Large Language Model","date":"2025-04-07","arxiv_id":"2504.04974","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-item-tokenization-for-transferable","title":"Universal Item Tokenization for Transferable Generative Recommendation","date":"2025-04-06","arxiv_id":"2504.04405","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-unified-referring-expression","title":"Towards Unified Referring Expression Segmentation Across Omni-Level Visual Target Granularities","date":"2025-04-02","arxiv_id":"2504.01954","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-qwen2vl-compute-efficient-pre-training","title":"Open-Qwen2VL: Compute-Efficient Pre-Training of Fully-Open Multimodal LLMs on Academic Resources","date":"2025-04-01","arxiv_id":"2504.00595","repositories_listed":0,"syntology":null},{"url":null,"slug":"orchmllm-orchestrate-multimodal-data-with","title":"Orchestrate Multimodal Data with Batch Post-Balancing to Accelerate Multimodal Large Language Model Training","date":"2025-03-31","arxiv_id":"2503.23830","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-pyramid-network-for-efficient","title":"Dynamic Pyramid Network for Efficient Multimodal Large Language Model","date":"2025-03-26","arxiv_id":"2503.20322","repositories_listed":0,"syntology":null},{"url":null,"slug":"mllm-for3d-adapting-multimodal-large-language","title":"MLLM-For3D: Adapting Multimodal Large Language Model for 3D Reasoning Segmentation","date":"2025-03-23","arxiv_id":"2503.18135","repositories_listed":0,"syntology":null},{"url":null,"slug":"legion-learning-to-ground-and-explain-for","title":"LEGION: Learning to Ground and Explain for Synthetic Image Detection","date":"2025-03-19","arxiv_id":"2503.15264","repositories_listed":0,"syntology":null},{"url":null,"slug":"upme-an-unsupervised-peer-review-framework","title":"UPME: An Unsupervised Peer Review Framework for Multimodal Large Language Model Evaluation","date":"2025-03-19","arxiv_id":"2503.14941","repositories_listed":0,"syntology":null},{"url":null,"slug":"spacevllm-endowing-multimodal-large-language","title":"SpaceVLLM: Endowing Multimodal Large Language Model with Spatio-Temporal Video Grounding Capability","date":"2025-03-18","arxiv_id":"2503.13983","repositories_listed":0,"syntology":null},{"url":null,"slug":"hide-llava-hierarchical-decoupling-for","title":"HiDe-LLaVA: Hierarchical Decoupling for Continual Instruction Tuning of Multimodal Large Language Model","date":"2025-03-17","arxiv_id":"2503.12941","repositories_listed":0,"syntology":null},{"url":null,"slug":"georsmllm-a-multimodal-large-language-model","title":"GeoRSMLLM: A Multimodal Large Language Model for Vision-Language Tasks in Geoscience and Remote Sensing","date":"2025-03-16","arxiv_id":"2503.12490","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-neural-implant-meets-multimodal-llm-a","title":"When neural implant meets multimodal LLM: A dual-loop system for neuromodulation and naturalistic neuralbehavioral research","date":"2025-03-16","arxiv_id":"2503.12334","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnidiff-a-comprehensive-benchmark-for-fine","title":"OmniDiff: A Comprehensive Benchmark for Fine-grained Image Difference Captioning","date":"2025-03-14","arxiv_id":"2503.11093","repositories_listed":0,"syntology":null},{"url":null,"slug":"cinema-coherent-multi-subject-video","title":"CINEMA: Coherent Multi-Subject Video Generation via MLLM-Based Guidance","date":"2025-03-13","arxiv_id":"2503.10391","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-agents-for-image-restoration","title":"Hybrid Agents for Image Restoration","date":"2025-03-13","arxiv_id":"2503.10120","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-multimodal-artificial","title":"Lightweight Multimodal Artificial Intelligence Framework for Maritime Multi-Scene Recognition","date":"2025-03-10","arxiv_id":"2503.06978","repositories_listed":0,"syntology":null},{"url":null,"slug":"cl-moe-enhancing-multimodal-large-language","title":"CL-MoE: Enhancing Multimodal Large Language Model with Dual Momentum Mixture-of-Experts for Continual Visual Question Answering","date":"2025-03-01","arxiv_id":"2503.00413","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimus-2-multimodal-minecraft-agent-with","title":"Optimus-2: Multimodal Minecraft Agent with Goal-Observation-Action Conditioned Policy","date":"2025-02-27","arxiv_id":"2502.19902","repositories_listed":0,"syntology":null},{"url":null,"slug":"gesture-aware-zero-shot-speech-recognition","title":"Gesture-Aware Zero-Shot Speech Recognition for Patients with Language Disorders","date":"2025-02-18","arxiv_id":"2502.13983","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmrc-a-large-scale-benchmark-for","title":"MMRC: A Large-Scale Benchmark for Understanding Multimodal Large Language Model in Real-World Conversation","date":"2025-02-17","arxiv_id":"2502.11903","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-multimodal-llms-assisted-by","title":"Leveraging Multimodal-LLMs Assisted by Instance Segmentation for Intelligent Traffic Monitoring","date":"2025-02-16","arxiv_id":"2502.11304","repositories_listed":0,"syntology":null},{"url":null,"slug":"distraction-is-all-you-need-for-multimodal","title":"Distraction is All You Need for Multimodal Large Language Model Jailbreaking","date":"2025-02-15","arxiv_id":"2502.10794","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-fairness-of-unified-multimodal-large","title":"On Fairness of Unified Multimodal Large Language Model for Image Generation","date":"2025-02-05","arxiv_id":"2502.03429","repositories_listed":0,"syntology":null},{"url":null,"slug":"mpic-position-independent-multimodal-context","title":"MPIC: Position-Independent Multimodal Context Caching System for Efficient MLLM Serving","date":"2025-02-04","arxiv_id":"2502.01960","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-free-token-reduction-for-multi-modal","title":"Learning Free Token Reduction for Multi-Modal Large Language Models","date":"2025-01-29","arxiv_id":"2501.17391","repositories_listed":0,"syntology":null},{"url":null,"slug":"humanomni-a-large-vision-speech-language","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","date":"2025-01-25","arxiv_id":"2501.15111","repositories_listed":0,"syntology":null},{"url":null,"slug":"eventvl-understand-event-streams-via","title":"EventVL: Understand Event Streams via Multimodal Large Language Model","date":"2025-01-23","arxiv_id":"2501.13707","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-droplet-digital-pcr-assay-for","title":"Interpretable Droplet Digital PCR Assay for Trustworthy Molecular Diagnostics","date":"2025-01-16","arxiv_id":"2501.09218","repositories_listed":0,"syntology":null},{"url":null,"slug":"omni-rgpt-unifying-image-and-video-region","title":"Omni-RGPT: Unifying Image and Video Region-level Understanding via Token Marks","date":"2025-01-14","arxiv_id":"2501.08326","repositories_listed":0,"syntology":null},{"url":null,"slug":"minmo-a-multimodal-large-language-model-for","title":"MinMo: A Multimodal Large Language Model for Seamless Voice Interaction","date":"2025-01-10","arxiv_id":"2501.06282","repositories_listed":0,"syntology":null},{"url":null,"slug":"llava-octopus-unlocking-instruction-driven","title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","date":"2025-01-09","arxiv_id":"2501.05067","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-face-anti-spoofing-enhancing","title":"Interpretable Face Anti-Spoofing: Enhancing Generalization with Multimodal Large Language Models","date":"2025-01-03","arxiv_id":"2501.01720","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-text-implementing-multimodal-large","title":"Beyond Text: Implementing Multimodal Large Language Model-Powered Multi-Agent Systems Using a No-Code Platform","date":"2025-01-01","arxiv_id":"2501.00750","repositories_listed":0,"syntology":null},{"url":null,"slug":"groundingface-fine-grained-face-understanding","title":"GroundingFace: Fine-grained Face Understanding via Pixel Grounding Multimodal Large Language Model","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"s4-driver-scalable-self-supervised-driving-1","title":"S4-Driver: Scalable Self-Supervised Driving Multimodal Large Language Model with Spatio-Temporal Visual Representation","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"st-3-accelerating-multimodal-large-language","title":"ST$^3$: Accelerating Multimodal Large Language Model by Spatial-Temporal Visual Token Trimming","date":"2024-12-28","arxiv_id":"2412.20105","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-large-scale-interpretable-multi-modality","title":"A Large-scale Interpretable Multi-modality Benchmark for Facial Image Forgery Localization","date":"2024-12-27","arxiv_id":"2412.19685","repositories_listed":0,"syntology":null},{"url":null,"slug":"substationai-multimodal-large-model-based","title":"SubstationAI: Multimodal Large Model-Based Approaches for Analyzing Substation Equipment Faults","date":"2024-12-22","arxiv_id":"2412.17077","repositories_listed":0,"syntology":null},{"url":null,"slug":"j-edi-qa-benchmark-for-deep-sea-organism","title":"J-EDI QA: Benchmark for deep-sea organism-specific multimodal LLM","date":"2024-12-20","arxiv_id":"2412.15574","repositories_listed":0,"syntology":null},{"url":null,"slug":"make-imagination-clearer-stable-diffusion","title":"Make Imagination Clearer! Stable Diffusion-based Visual Imagination for Multimodal Machine Translation","date":"2024-12-17","arxiv_id":"2412.12627","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-mathematical-reasoning-in-the-era","title":"A Survey of Mathematical Reasoning in the Era of Multimodal Large Language Model: Benchmark, Method & Challenges","date":"2024-12-16","arxiv_id":"2412.11936","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-speech-foundation-model-for","title":"MERaLiON-SpeechEncoder: Towards a Speech Foundation Model for Singapore and Beyond","date":"2024-12-16","arxiv_id":"2412.11538","repositories_listed":0,"syntology":null},{"url":null,"slug":"easyref-omni-generalized-group-image","title":"EasyRef: Omni-Generalized Group Image Reference for Diffusion Models via Multimodal LLM","date":"2024-12-12","arxiv_id":"2412.09618","repositories_listed":0,"syntology":null},{"url":null,"slug":"coef-vq-cost-efficient-video-quality","title":"COEF-VQ: Cost-Efficient Video Quality Understanding through a Cascaded Multimodal LLM Framework","date":"2024-12-11","arxiv_id":"2412.10435","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffsensei-bridging-multi-modal-llms-and","title":"DiffSensei: Bridging Multi-Modal LLMs and Diffusion Models for Customized Manga Generation","date":"2024-12-10","arxiv_id":"2412.07589","repositories_listed":0,"syntology":null},{"url":"/paper/illume-illuminating-your-llms-to-see-draw-and","slug":"illume-illuminating-your-llms-to-see-draw-and","title":"ILLUME: Illuminating Your LLMs to See, Draw, and Self-Enhance","date":"2024-12-09","arxiv_id":"2412.06673","repositories_listed":0,"syntology":null},{"url":null,"slug":"editscout-locating-forged-regions-from","title":"EditScout: Locating Forged Regions from Diffusion-based Edited Images with Multimodal LLM","date":"2024-12-05","arxiv_id":"2412.03809","repositories_listed":0,"syntology":null},{"url":null,"slug":"egoplan-bench2-a-benchmark-for-multimodal","title":"EgoPlan-Bench2: A Benchmark for Multimodal Large Language Model Planning in Real-World Scenarios","date":"2024-12-05","arxiv_id":"2412.04447","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamiccontrol-adaptive-condition-selection","title":"DynamicControl: Adaptive Condition Selection for Improved Text-to-Image Generation","date":"2024-12-04","arxiv_id":"2412.03255","repositories_listed":0,"syntology":null},{"url":null,"slug":"objectfinder-open-vocabulary-assistive-system","title":"ObjectFinder: An Open-Vocabulary Assistive System for Interactive Object Search by Blind People","date":"2024-12-04","arxiv_id":"2412.03118","repositories_listed":0,"syntology":null},{"url":null,"slug":"wsi-llava-a-multimodal-large-language-model","title":"WSI-LLaVA: A Multimodal Large Language Model for Whole Slide Image","date":"2024-12-03","arxiv_id":"2412.02141","repositories_listed":0,"syntology":null},{"url":null,"slug":"motrans-customized-motion-transfer-with-text","title":"MoTrans: Customized Motion Transfer with Text-driven Video Diffusion Models","date":"2024-12-02","arxiv_id":"2412.01343","repositories_listed":0,"syntology":null},{"url":null,"slug":"seqafford-sequential-3d-affordance-reasoning","title":"SeqAfford: Sequential 3D Affordance Reasoning via Multimodal Large Language Model","date":"2024-12-02","arxiv_id":"2412.01550","repositories_listed":0,"syntology":null},{"url":null,"slug":"realistic-corner-case-generation-for","title":"Realistic Corner Case Generation for Autonomous Vehicles with Multimodal Large Language Model","date":"2024-11-29","arxiv_id":"2412.00243","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-large-language-model-for-wheat","title":"Multimodal large language model for wheat breeding: a new exploration of smart breeding","date":"2024-11-20","arxiv_id":"2411.15203","repositories_listed":0,"syntology":null},{"url":null,"slug":"cue-m-contextual-understanding-and-enhanced","title":"CUE-M: Contextual Understanding and Enhanced Search with Multimodal Large Language Model","date":"2024-11-19","arxiv_id":"2411.12287","repositories_listed":0,"syntology":null},{"url":null,"slug":"med-2e3-a-2d-enhanced-3d-medical-multimodal","title":"Med-2E3: A 2D-Enhanced 3D Medical Multimodal Large Language Model","date":"2024-11-19","arxiv_id":"2411.12783","repositories_listed":0,"syntology":null},{"url":null,"slug":"streetviewllm-extracting-geographic","title":"StreetviewLLM: Extracting Geographic Information Using a Chain-of-Thought Multimodal Large Language Model","date":"2024-11-19","arxiv_id":"2411.14476","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-hallucination-in-multimodal-large","title":"Mitigating Hallucination in Multimodal Large Language Model via Hallucination-targeted Direct Preference Optimization","date":"2024-11-15","arxiv_id":"2411.10436","repositories_listed":0,"syntology":null},{"url":"/paper/capellm-support-free-category-agnostic-pose","slug":"capellm-support-free-category-agnostic-pose","title":"CapeLLM: Support-Free Category-Agnostic Pose Estimation with Multimodal Large Language Models","date":"2024-11-11","arxiv_id":"2411.06869","repositories_listed":0,"syntology":null},{"url":null,"slug":"toursynbio-search-a-large-language-model","title":"TourSynbio-Search: A Large Language Model Driven Agent Framework for Unified Search Method for Protein Engineering","date":"2024-11-09","arxiv_id":"2411.06024","repositories_listed":0,"syntology":null},{"url":null,"slug":"chattracker-enhancing-visual-tracking","title":"ChatTracker: Enhancing Visual Tracking Performance via Chatting with Multimodal Large Language Model","date":"2024-11-04","arxiv_id":"2411.01756","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-multimodal-large-language-model-think","title":"Can Multimodal Large Language Model Think Analogically?","date":"2024-11-02","arxiv_id":"2411.01307","repositories_listed":0,"syntology":null},{"url":null,"slug":"web-scale-visual-entity-recognition-an-llm","title":"Web-Scale Visual Entity Recognition: An LLM-Driven Data Approach","date":"2024-10-31","arxiv_id":"2410.23676","repositories_listed":0,"syntology":null},{"url":null,"slug":"ferret-ui-2-mastering-universal-user","title":"Ferret-UI 2: Mastering Universal User Interface Understanding Across Platforms","date":"2024-10-24","arxiv_id":"2410.18967","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-bilingual-multimodal-large","title":"Interpretable Bilingual Multimodal Large Language Model for Diverse Biomedical Tasks","date":"2024-10-24","arxiv_id":"2410.18387","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-real-zero-shot-camouflaged-object","title":"Towards Real Zero-Shot Camouflaged Object Segmentation without Camouflaged Annotations","date":"2024-10-22","arxiv_id":"2410.16953","repositories_listed":0,"syntology":null},{"url":null,"slug":"llava-ultra-large-chinese-language-and-vision","title":"LLaVA-Ultra: Large Chinese Language and Vision Assistant for Ultrasound","date":"2024-10-19","arxiv_id":"2410.15074","repositories_listed":0,"syntology":null},{"url":null,"slug":"mochat-joints-grouped-spatio-temporal","title":"MoChat: Joints-Grouped Spatio-Temporal Grounding LLM for Multi-Turn Motion Comprehension and Description","date":"2024-10-15","arxiv_id":"2410.11404","repositories_listed":0,"syntology":null},{"url":null,"slug":"forgerygpt-multimodal-large-language-model","title":"ForgeryGPT: Multimodal Large Language Model For Explainable Image Forgery Detection and Localization","date":"2024-10-14","arxiv_id":"2410.10238","repositories_listed":0,"syntology":null},{"url":null,"slug":"vit3d-alignment-of-llama3-3d-medical-image","title":"ViT3D Alignment of LLaMA3: 3D Medical Image Report Generation","date":"2024-10-11","arxiv_id":"2410.08588","repositories_listed":0,"syntology":null},{"url":null,"slug":"respllm-unifying-audio-and-text-with","title":"RespLLM: Unifying Audio and Text with Multimodal LLMs for Generalized Respiratory Health Prediction","date":"2024-10-07","arxiv_id":"2410.05361","repositories_listed":0,"syntology":null},{"url":null,"slug":"occ-mllm-empowering-multimodal-large-language","title":"OCC-MLLM:Empowering Multimodal Large Language Model For the Understanding of Occluded Objects","date":"2024-10-02","arxiv_id":"2410.01261","repositories_listed":0,"syntology":null},{"url":null,"slug":"vmad-visual-enhanced-multimodal-large","title":"VMAD: Visual-enhanced Multimodal Large Language Model for Zero-Shot Anomaly Detection","date":"2024-09-30","arxiv_id":"2409.20146","repositories_listed":0,"syntology":null},{"url":null,"slug":"cadvlm-bridging-language-and-vision-in-the","title":"CadVLM: Bridging Language and Vision in the Generation of Parametric CAD Sketches","date":"2024-09-26","arxiv_id":"2409.17457","repositories_listed":0,"syntology":null},{"url":null,"slug":"eagle-egocentric-aggregated-language-video","title":"EAGLE: Egocentric AGgregated Language-video Engine","date":"2024-09-26","arxiv_id":"2409.17523","repositories_listed":0,"syntology":null},{"url":null,"slug":"clsp-high-fidelity-contrastive-language-state","title":"CLSP: High-Fidelity Contrastive Language-State Pre-training for Agent State Representation","date":"2024-09-24","arxiv_id":"2409.15806","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-style-efficient-fine-tuning-of-llms","title":"Decoding Style: Efficient Fine-Tuning of LLMs for Image-Guided Outfit Recommendation with Preference","date":"2024-09-18","arxiv_id":"2409.12150","repositories_listed":0,"syntology":null},{"url":null,"slug":"mip-gaf-a-mllm-annotated-benchmark-for-most","title":"MIP-GAF: A MLLM-annotated Benchmark for Most Important Person Localization and Group Context Understanding","date":"2024-09-10","arxiv_id":"2409.06224","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-large-language-model-driven","title":"Multimodal Large Language Model Driven Scenario Testing for Autonomous Vehicles","date":"2024-09-10","arxiv_id":"2409.06450","repositories_listed":0,"syntology":null},{"url":null,"slug":"mllm-fl-multimodal-large-language-model","title":"MLLM-LLaVA-FL: Multimodal Large Language Model Assisted Federated Learning","date":"2024-09-09","arxiv_id":"2409.06067","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-medical-multimodal-large-language-model-for","title":"A Medical Multimodal Large Language Model for Pediatric Pneumonia","date":"2024-09-04","arxiv_id":"2409.02608","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-performance-and-efficiency-a","title":"Balancing Performance and Efficiency: A Multimodal Large Language Model Pruning Method based Image Text Interaction","date":"2024-09-02","arxiv_id":"2409.01162","repositories_listed":0,"syntology":null},{"url":null,"slug":"dpdedit-detail-preserved-diffusion-models-for","title":"DPDEdit: Detail-Preserved Diffusion Models for Multimodal Fashion Image Editing","date":"2024-09-02","arxiv_id":"2409.01086","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-multi-turn-conversation-stance","title":"Multimodal Multi-turn Conversation Stance Detection: A Challenge Dataset and Effective Model","date":"2024-09-01","arxiv_id":"2409.00597","repositories_listed":0,"syntology":null},{"url":null,"slug":"orthodoc-multimodal-large-language-model-for","title":"OrthoDoc: Multimodal Large Language Model for Assisting Diagnosis in Computed Tomography","date":"2024-08-30","arxiv_id":"2409.09052","repositories_listed":0,"syntology":null},{"url":"/paper/maven-an-effective-multi-granularity-hybrid","slug":"maven-an-effective-multi-granularity-hybrid","title":"MaVEn: An Effective Multi-granularity Hybrid Visual Encoding Framework for Multimodal Large Language Model","date":"2024-08-22","arxiv_id":"2408.12321","repositories_listed":0,"syntology":null},{"url":null,"slug":"vintern-1b-an-efficient-multimodal-large","title":"Vintern-1B: An Efficient Multimodal Large Language Model for Vietnamese","date":"2024-08-22","arxiv_id":"2408.12480","repositories_listed":0,"syntology":null},{"url":null,"slug":"cardiff-video-salient-object-ranking-chain-of","title":"CaRDiff: Video Salient Object Ranking Chain of Thought Reasoning for Saliency Prediction with Diffusion","date":"2024-08-21","arxiv_id":"2408.12009","repositories_listed":0,"syntology":null},{"url":null,"slug":"ee-mllm-a-data-efficient-and-compute","title":"EE-MLLM: A Data-Efficient and Compute-Efficient Multimodal Large Language Model","date":"2024-08-21","arxiv_id":"2408.11795","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-emotion-open-vocabulary-recognition","title":"Video Emotion Open-vocabulary Recognition Based on Multimodal Large Language Model","date":"2024-08-21","arxiv_id":"2408.11286","repositories_listed":0,"syntology":null},{"url":null,"slug":"panosent-a-panoptic-sextuple-extraction","title":"PanoSent: A Panoptic Sextuple Extraction Benchmark for Multimodal Conversational Aspect-based Sentiment Analysis","date":"2024-08-18","arxiv_id":"2408.09481","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatgpt-meets-iris-biometrics","title":"ChatGPT Meets Iris Biometrics","date":"2024-08-09","arxiv_id":"2408.04868","repositories_listed":0,"syntology":null}],"record_sha256":"2429ecac3b821144e141167a6b8f0bf8be866fab8ca0466005ef5a0c5bc93de8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}