{"url":"/dataset/sme","name":"SME","full_name":"Standard Multimodal Explanation","description_markdown":"SME is a new dataset for Multi-modal Explanation for Visual Question Answering comprising 1,028,230 samples, with 1,656 visual objects requiring detection in explanations. To our knowledge, this is the first dataset where the explanations are in standard English with additional visual grounding tokens.","description_withheld":null,"homepage":"https://huggingface.co/datasets/LivXue/SME","introduced_date":"2024-10-28","introduced_date_note":null,"introduced_by":{"paper":"/paper/few-shot-multimodal-explanation-for-visual","title":"Few-Shot Multimodal Explanation for Visual Question Answering","first_author":"Dizhan Xue","url":null},"license":{"name":"Apache-2.0","url":"https://github.com/LivXue/FS-MEVQA/blob/main/LICENSE"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Explanatory Visual Question Answering","url":"/task/explanatory-visual-question-answering","datasets_with_task":"/datasets/task/explanatory-visual-question-answering"},{"name":"FS-MEVQA","url":"/task/fs-mevqa","datasets_with_task":"/datasets/task/fs-mevqa"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["SME"],"data_loaders":[{"repo":"https://github.com/LivXue/FS-MEVQA","url":"https://github.com/LivXue/FS-MEVQA","frameworks":["pytorch"]}],"num_papers_in_archive":7,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/fs-mevqa-on-sme","task":"FS-MEVQA","dataset_variant":"SME","rows":7,"metrics":["BLEU-4","METEOR","ROUGE-L","CIDEr","SPICE","Detection","ACC","#Learning Samples (N)"],"first_row_in_archive_order":{"model":"MEAgent","paper":"/paper/few-shot-multimodal-explanation-for-visual","metrics":{"#Learning Samples (N)":"16","ACC":"51.45","BLEU-4":"67.91","CIDEr":"510.44","Detection":"29.09","METEOR":"50.55","ROUGE-L":"79.41","SPICE":"64.09"},"code_links":[{"title":"LivXue/FS-MEVQA","url":"https://github.com/LivXue/FS-MEVQA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/few-shot-multimodal-explanation-for-visual","title":"Few-Shot Multimodal Explanation for Visual Question Answering","date":"2024-10-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/gemini-1-5-unlocking-multimodal-understanding","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","date":"2024-03-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/cogvlm-visual-expert-for-pretrained-language","title":"CogVLM: Visual Expert for Pretrained Language Models","date":"2023-11-06","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/qwen-vl-a-frontier-large-vision-language","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","date":"2023-08-24","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/variational-causal-inference-network-for","title":"Variational Causal Inference Network for Explanatory Visual Question Answering","date":"2023-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rex-reasoning-aware-and-grounded-explanation","title":"REX: Reasoning-aware and Grounded Explanation","date":"2022-03-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":3,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}