{"url":"/task/mme","name":"MME","slug":"mme","description_markdown":"MME is a comprehensive evaluation benchmark for multimodal large language models. It measures both perception and cognition abilities on a total of 14 subtasks, including existence, count, position, color, poster, celebrity, scene, landmark, artwork, OCR, commonsense reasoning, numerical calculation, text translation, and code reasoning.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Reasoning","url":"/area/reasoning"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":95,"papers_with_code":47,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":0,"parent_tasks":1},"benchmarks":[],"datasets":[],"subtasks":[],"parent_tasks":[{"url":"/task/multimodal-reasoning","name":"Multimodal Reasoning"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":47,"tagged_in_all":95,"items":[{"url":"/paper/semi-supervised-domain-adaptation-via-minimax","title":"Semi-supervised Domain Adaptation via Minimax Entropy","date":"2019-04-13","arxiv_id":"1904.06487","repositories_listed":5,"syntology":{"n":24,"n_ran":5,"n_unverified":19,"n_pointer_only":2}},{"url":"/paper/mme-survey-a-comprehensive-survey-on","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","date":"2024-11-22","arxiv_id":"2411.15296","repositories_listed":4,"syntology":null},{"url":"/paper/mme-a-comprehensive-evaluation-benchmark-for","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","date":"2023-06-23","arxiv_id":"2306.13394","repositories_listed":4,"syntology":null},{"url":"/paper/internlm-xcomposer-a-vision-language-large","title":"InternLM-XComposer: A Vision-Language Large Model for Advanced Text-image Comprehension and Composition","date":"2023-09-26","arxiv_id":"2309.15112","repositories_listed":3,"syntology":null},{"url":"/paper/videoeval-pro-robust-and-realistic-long-video","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","date":"2025-05-20","arxiv_id":"2505.14640","repositories_listed":2,"syntology":{"n":14,"n_ran":5,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/timechat-online-80-visual-tokens-are","title":"TimeChat-Online: 80% Visual Tokens are Naturally Redundant in Streaming Videos","date":"2025-04-24","arxiv_id":"2504.17343","repositories_listed":2,"syntology":{"n":12,"n_ran":7,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/spatial-r1-enhancing-mllms-in-video-spatial","title":"SpaceR: Reinforcing MLLMs in Video Spatial Reasoning","date":"2025-04-02","arxiv_id":"2504.01805","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/long-context-transfer-from-language-to-vision","title":"Long Context Transfer from Language to Vision","date":"2024-06-24","arxiv_id":"2406.16852","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":5}},{"url":"/paper/mitigating-hallucinations-in-large-vision","title":"Mitigating Hallucinations in Large Vision-Language Models with Instruction Contrastive Decoding","date":"2024-03-27","arxiv_id":"2403.18715","repositories_listed":2,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":1}},{"url":"/paper/a-challenger-to-gpt-4v-early-explorations-of","title":"A Challenger to GPT-4V? Early Explorations of Gemini in Visual Expertise","date":"2023-12-19","arxiv_id":"2312.12436","repositories_listed":2,"syntology":null},{"url":"/paper/mmicl-empowering-vision-language-model-with","title":"MMICL: Empowering Vision-language Model with Multi-Modal In-Context Learning","date":"2023-09-14","arxiv_id":"2309.07915","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":7}},{"url":"/paper/m-3-video-masked-motion-modeling-for-self","title":"Masked Motion Encoding for Self-Supervised Video Representation Learning","date":"2022-10-12","arxiv_id":"2210.06096","repositories_listed":2,"syntology":null},{"url":"/paper/high-resolution-visual-reasoning-via-multi","title":"High-Resolution Visual Reasoning via Multi-Turn Grounding-Based Reinforcement Learning","date":"2025-07-08","arxiv_id":"2507.05920","repositories_listed":1,"syntology":null},{"url":"/paper/flash-vstream-efficient-real-time","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","date":"2025-06-30","arxiv_id":"2506.23825","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/videodeepresearch-long-video-understanding","title":"VideoDeepResearch: Long Video Understanding With Agentic Tool Using","date":"2025-06-12","arxiv_id":"2506.10821","repositories_listed":1,"syntology":null},{"url":"/paper/silvr-a-simple-language-based-video-reasoning","title":"SiLVR: A Simple Language-based Video Reasoning Framework","date":"2025-05-30","arxiv_id":"2505.24869","repositories_listed":1,"syntology":null},{"url":"/paper/frag-frame-selection-augmented-generation-for","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","date":"2025-04-24","arxiv_id":"2504.17447","repositories_listed":1,"syntology":null},{"url":"/paper/eagle-2-5-boosting-long-context-post-training","title":"Eagle 2.5: Boosting Long-Context Post-Training for Frontier Vision-Language Models","date":"2025-04-21","arxiv_id":"2504.15271","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/bolt-boost-large-vision-language-model","title":"BOLT: Boost Large Vision-Language Model Without Training for Long-form Video Understanding","date":"2025-03-27","arxiv_id":"2503.21483","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/instruction-aligned-visual-attention-for","title":"Instruction-Aligned Visual Attention for Mitigating Hallucinations in Large Vision-Language Models","date":"2025-03-24","arxiv_id":"2503.18556","repositories_listed":1,"syntology":null},{"url":"/paper/quota-query-oriented-token-assignment-via-cot","title":"QuoTA: Query-oriented Token Assignment via CoT Query Decouple for Long Video Comprehension","date":"2025-03-11","arxiv_id":"2503.08689","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/re-imagining-multimodal-instruction-tuning-a","title":"Re-Imagining Multimodal Instruction Tuning: A Representation View","date":"2025-03-02","arxiv_id":"2503.00723","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":4}},{"url":"/paper/towards-text-image-interleaved-retrieval","title":"Towards Text-Image Interleaved Retrieval","date":"2025-02-18","arxiv_id":"2502.12799","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/expand-vsr-benchmark-for-vllm-to-expertize-in","title":"Expand VSR Benchmark for VLLM to Expertize in Spatial Rules","date":"2024-12-24","arxiv_id":"2412.18224","repositories_listed":1,"syntology":null},{"url":"/paper/lyra-an-efficient-and-speech-centric","title":"Lyra: An Efficient and Speech-Centric Framework for Omni-Cognition","date":"2024-12-12","arxiv_id":"2412.09501","repositories_listed":1,"syntology":{"n":19,"n_ran":3,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/video-rag-visually-aligned-retrieval","title":"Video-RAG: Visually-aligned Retrieval-Augmented Long Video Comprehension","date":"2024-11-20","arxiv_id":"2411.13093","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/dynamic-multimodal-evaluation-with-flexible","title":"Dynamic Multimodal Evaluation with Flexible Complexity by Vision-Language Bootstrapping","date":"2024-10-11","arxiv_id":"2410.08695","repositories_listed":1,"syntology":null},{"url":"/paper/to-preserve-or-to-compress-an-in-depth-study","title":"To Preserve or To Compress: An In-Depth Study of Connector Selection in Multimodal Large Language Models","date":"2024-10-09","arxiv_id":"2410.06765","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-modality-prior-induced","title":"Mitigating Modality Prior-Induced Hallucinations in Multimodal Large Language Models via Deciphering Attention Causality","date":"2024-10-07","arxiv_id":"2410.04780","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/tubench-benchmarking-large-vision-language","title":"TUBench: Benchmarking Large Vision-Language Models on Trustworthiness with Unanswerable Questions","date":"2024-10-05","arxiv_id":"2410.04107","repositories_listed":1,"syntology":null}],"syntology_records":16,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}