{"url":"/task/multimodal-large-language-model","name":"Multimodal Large Language Model","slug":"multimodal-large-language-model","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":347,"papers_with_code":160,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/offensive-memes-in-singapore-context","name":"Offensive Memes in Singapore Context","full_name":"Offensive Memes in Singapore Context","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":160,"tagged_in_all":347,"items":[{"url":"/paper/mme-a-comprehensive-evaluation-benchmark-for","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","date":"2023-06-23","arxiv_id":"2306.13394","repositories_listed":4,"syntology":null},{"url":"/paper/shapellm-universal-3d-object-understanding","title":"ShapeLLM: Universal 3D Object Understanding for Embodied Interaction","date":"2024-02-27","arxiv_id":"2402.17766","repositories_listed":3,"syntology":{"n":17,"n_ran":9,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/vargpt-unified-understanding-and-generation","title":"VARGPT: Unified Understanding and Generation in a Visual Autoregressive Multimodal Large Language Model","date":"2025-01-21","arxiv_id":"2501.12327","repositories_listed":2,"syntology":null},{"url":"/paper/remote-sensing-temporal-vision-language","title":"Remote Sensing Temporal Vision-Language Models: A Comprehensive Survey","date":"2024-12-03","arxiv_id":"2412.02573","repositories_listed":2,"syntology":null},{"url":"/paper/learn-from-downstream-and-be-yourself-in","title":"Learn from Downstream and Be Yourself in Multimodal Large Language Model Fine-Tuning","date":"2024-11-17","arxiv_id":"2411.10928","repositories_listed":2,"syntology":null},{"url":"/paper/baichuan-omni-technical-report","title":"Baichuan-Omni Technical Report","date":"2024-10-11","arxiv_id":"2410.08565","repositories_listed":2,"syntology":null},{"url":"/paper/ovis-structural-embedding-alignment-for","title":"Ovis: Structural Embedding Alignment for Multimodal Large Language Model","date":"2024-05-31","arxiv_id":"2405.20797","repositories_listed":2,"syntology":null},{"url":"/paper/minigpt4-video-advancing-multimodal-llms-for","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","date":"2024-04-04","arxiv_id":"2404.03413","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/jailbreaking-attack-against-multimodal-large","title":"Jailbreaking Attack against Multimodal Large Language Model","date":"2024-02-04","arxiv_id":"2402.02309","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/tool-lmm-a-large-multi-modal-model-for-tool","title":"MLLM-Tool: A Multimodal Large Language Model For Tool Agent Learning","date":"2024-01-19","arxiv_id":"2401.10727","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/tinygpt-v-efficient-multimodal-large-language","title":"TinyGPT-V: Efficient Multimodal Large Language Model via Small Backbones","date":"2023-12-28","arxiv_id":"2312.16862","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/timechat-a-time-sensitive-multimodal-large","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","date":"2023-12-04","arxiv_id":"2312.02051","repositories_listed":2,"syntology":{"n":11,"n_ran":7,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/ferret-refer-and-ground-anything-anywhere-at","title":"Ferret: Refer and Ground Anything Anywhere at Any Granularity","date":"2023-10-11","arxiv_id":"2310.07704","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_unverified":1,"n_pointer_only":8}},{"url":"/paper/kosmos-2-grounding-multimodal-large-language","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","date":"2023-06-26","arxiv_id":"2306.14824","repositories_listed":2,"syntology":null},{"url":"/paper/mfgdiffusion-mask-guided-smoke-synthesis-for","title":"MFGDiffusion: Mask-Guided Smoke Synthesis for Enhanced Forest Fire Detection","date":"2025-07-15","arxiv_id":"2507.11252","repositories_listed":1,"syntology":null},{"url":"/paper/thinksound-chain-of-thought-reasoning-in","title":"ThinkSound: Chain-of-Thought Reasoning in Multimodal Large Language Models for Audio Generation and Editing","date":"2025-06-26","arxiv_id":"2506.21448","repositories_listed":1,"syntology":null},{"url":"/paper/oraclefusion-assisting-the-decipherment-of","title":"OracleFusion: Assisting the Decipherment of Oracle Bone Script with Structurally Constrained Semantic Typography","date":"2025-06-26","arxiv_id":"2506.21101","repositories_listed":1,"syntology":null},{"url":"/paper/medtvt-r1-a-multimodal-llm-empowering-medical","title":"MedTVT-R1: A Multimodal LLM Empowering Medical Reasoning and Diagnosis","date":"2025-06-23","arxiv_id":"2506.18512","repositories_listed":1,"syntology":null},{"url":"/paper/sharegpt-4o-image-aligning-multimodal-models","title":"ShareGPT-4o-Image: Aligning Multimodal Models with GPT-4o-Level Image Generation","date":"2025-06-22","arxiv_id":"2506.18095","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/the-condition-number-as-a-scale-invariant","title":"The Condition Number as a Scale-Invariant Proxy for Information Encoding in Neural Units","date":"2025-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/vis-shepherd-constructing-critic-for-llm","title":"VIS-Shepherd: Constructing Critic for LLM-based Data Visualization Generation","date":"2025-06-16","arxiv_id":"2506.13326","repositories_listed":1,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":6}},{"url":"/paper/period-llm-extending-the-periodic-capability","title":"Period-LLM: Extending the Periodic Capability of Multimodal Large Language Model","date":"2025-05-30","arxiv_id":"2505.24476","repositories_listed":1,"syntology":null},{"url":"/paper/un-2-clip-improving-clip-s-visual-detail","title":"un$^2$CLIP: Improving CLIP's Visual Detail Capturing Ability via Inverting unCLIP","date":"2025-05-30","arxiv_id":"2505.24517","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-rag-sub-dimensional-retrieval","title":"Cross-modal RAG: Sub-dimensional Retrieval-Augmented Text-to-Image Generation","date":"2025-05-28","arxiv_id":"2505.21956","repositories_listed":1,"syntology":null},{"url":"/paper/geollava-8k-scaling-remote-sensing-multimodal","title":"GeoLLaVA-8K: Scaling Remote-Sensing Multimodal Large Language Models to 8K Resolution","date":"2025-05-27","arxiv_id":"2505.21375","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-llm-guided-semantic-correction-in","title":"Multimodal LLM-Guided Semantic Correction in Text-to-Image Diffusion","date":"2025-05-26","arxiv_id":"2505.20053","repositories_listed":1,"syntology":null},{"url":"/paper/diagnosing-and-mitigating-modality","title":"Diagnosing and Mitigating Modality Interference in Multimodal Large Language Models","date":"2025-05-26","arxiv_id":"2505.19616","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/unifying-multimodal-large-language-model","title":"Unifying Multimodal Large Language Model Capabilities and Modalities via Model Merging","date":"2025-05-26","arxiv_id":"2505.19892","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":5}},{"url":"/paper/chemmllm-chemical-multimodal-large-language","title":"ChemMLLM: Chemical Multimodal Large Language Model","date":"2025-05-22","arxiv_id":"2505.16326","repositories_listed":1,"syntology":null},{"url":"/paper/dimple-discrete-diffusion-multimodal-large","title":"Dimple: Discrete Diffusion Multimodal Large Language Model with Parallel Decoding","date":"2025-05-22","arxiv_id":"2505.16990","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}}],"syntology_records":12,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}