{"url":"/dataset/mmbench","name":"MMBench","full_name":null,"description_markdown":"**MMBench** is a multi-modality benchmark. It methodically develops a comprehensive evaluation pipeline, primarily comprised of two elements. The first element is a meticulously curated dataset that surpasses existing similar benchmarks in terms of the number and variety of evaluation questions and abilities. The second element introduces a novel CircularEval strategy and incorporates the use of ChatGPT. This implementation is designed to convert free-form predictions into pre-defined choices, thereby facilitating a more robust evaluation of the model's predictions.","description_withheld":null,"homepage":"https://opencompass.org.cn/mmbench","introduced_date":"2023-07-12","introduced_date_note":null,"introduced_by":{"paper":"/paper/mmbench-is-your-multi-modal-model-an-all","title":"MMBench: Is Your Multi-modal Model an All-around Player?","first_author":"YuAn Liu","url":null},"license":{"name":"Apache-2.0 license","url":"https://github.com/InternLM/opencompass/blob/main/LICENSE"},"modalities":[],"tasks":[{"name":"Visual Question Answering","url":"/task/visual-question-answering-1","datasets_with_task":"/datasets/task/visual-question-answering-1"}],"languages":[],"variants":["MMBench"],"data_loaders":[],"num_papers_in_archive":384,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-question-answering-on-mmbench","task":"Visual Question Answering","dataset_variant":"MMBench","rows":5,"metrics":["GPT-3.5 score"],"first_row_in_archive_order":{"model":"LLaVA-InternLM2-ViT + MoSLoRA","paper":"/paper/mixture-of-subspaces-in-low-rank-adaptation","metrics":{"GPT-3.5 score":"73.8"},"code_links":[{"title":"wutaiqiang/moslora","url":"https://github.com/wutaiqiang/moslora"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/mixture-of-subspaces-in-low-rank-adaptation","title":"Mixture-of-Subspaces in Low-Rank Adaptation","date":"2024-06-16","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","date":"2024-05-09","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":10,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-lavit-unified-video-language-pre","title":"Video-LaVIT: Unified Video-Language Pre-training with Decoupled Visual-Motional Tokenization","date":"2024-02-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dreamllm-synergistic-multimodal-comprehension","title":"DreamLLM: Synergistic Multimodal Comprehension and Creation","date":"2023-09-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":26,"samples_ran":20,"samples_unverified":6,"pointer_only_for_licence":13,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}