{"url":"/dataset/clevr-x","name":"CLEVR-X","full_name":null,"description_markdown":"**CLEVR-X** is a dataset that extends the [CLEVR](/dataset/clevr) dataset with natural language explanations in the context of  VQA. It consists of 3.6 million natural language explanations for 850k question-image pairs.\r\n\r\nFor each image-question pair in the CLEVR dataset, CLEVR-X contains multiple structured textual explanations which are derived from the original scene graphs. By construction, the CLEVR-X explanations are correct and describe the reasoning and visual information that is necessary to answer a given question.\r\n\r\nThe CLEVR-X dataset consists of:\r\n\r\n- A training set of 2,401,275 natural language explanations for 70,000 images.\r\n- A validation set of 599,711 natural language explanations for 14,000 images.\r\n- A test set of 644,151 natural language explanations for 15,000 images.","description_withheld":null,"homepage":"https://explainableml.github.io/CLEVR-X/","introduced_date":"2022-04-05","introduced_date_note":null,"introduced_by":{"paper":"/paper/clevr-x-a-visual-reasoning-dataset-for","title":"CLEVR-X: A Visual Reasoning Dataset for Natural Language Explanations","first_author":"Leonard Salewski","url":null},"license":{"name":"BSD-3-Clause License","url":"https://github.com/ExplainableML/CLEVR-X/blob/master/LICENSE"},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Explanation Generation","url":"/task/explanation-generation","datasets_with_task":"/datasets/task/explanation-generation"}],"languages":[],"variants":["CLEVR-X"],"data_loaders":[{"repo":"https://github.com/explainableml/clevr-x","url":"https://github.com/explainableml/clevr-x","frameworks":[]}],"num_papers_in_archive":5,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/explanation-generation-on-clevr-x","task":"Explanation Generation","dataset_variant":"CLEVR-X","rows":2,"metrics":["B4","M","RL","C","Acc"],"first_row_in_archive_order":{"model":"PJ-X","paper":"/paper/clevr-x-a-visual-reasoning-dataset-for","metrics":{"Acc":"63.0","B4":"87.4","C":"639.8","M":"58.9","RL":"93.4"},"code_links":[{"title":"explainableml/clevr-x","url":"https://github.com/explainableml/clevr-x"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/clevr-x-a-visual-reasoning-dataset-for","title":"CLEVR-X: A Visual Reasoning Dataset for Natural Language Explanations","date":"2022-04-05","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}