{"url":"/dataset/llava-bench-in-the-wild","name":"LLaVA-Bench","full_name":null,"description_markdown":"LLaVA-Bench is a dataset created to evaluate the capability of large multimodal models (LMM) in more challenging tasks and generalizability to novel domains. It consists of a diverse set of 24 images with 60 questions in total, including indoor and outdoor scenes, memes, paintings, sketches, etc., and each image with a highly-detailed and manually-curated description and a proper selection of questions. The dataset is part of the LLaVA project, which aims to develop multimodal chatbots that follow human intents to complete various daily-life visual tasks in the wild.","description_withheld":null,"homepage":"https://llava-vl.github.io/","introduced_date":"2023-04-17","introduced_date_note":null,"introduced_by":{"paper":"/paper/visual-instruction-tuning-1","title":"Visual Instruction Tuning","first_author":"Haotian Liu","url":null},"license":null,"modalities":[],"tasks":[{"name":"visual instruction following","url":"/task/visual-instruction-following","datasets_with_task":"/datasets/task/visual-instruction-following"}],"languages":[],"variants":["LLaVA-Bench"],"data_loaders":[],"num_papers_in_archive":130,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-instruction-following-on-llava-bench","task":"visual instruction following","dataset_variant":"LLaVA-Bench","rows":8,"metrics":["avg score"],"first_row_in_archive_order":{"model":"CuMo-7B","paper":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","metrics":{"avg score":"85.7"},"code_links":[{"title":"shi-labs/cumo","url":"https://github.com/shi-labs/cumo"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","date":"2024-05-09","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":10,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sharegpt4v-improving-large-multi-modal-models","title":"ShareGPT4V: Improving Large Multi-Modal Models with Better Captions","date":"2023-11-21","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/improved-baselines-with-visual-instruction","title":"Improved Baselines with Visual Instruction Tuning","date":"2023-10-05","rows_on_this_dataset":2,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":6,"samples_unverified":3,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/instructblip-towards-general-purpose-vision","title":"InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning","date":"2023-05-11","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","rows_on_this_dataset":1,"code_links":17,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":29,"samples_ran":20,"samples_unverified":9,"pointer_only_for_licence":9,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}