{"url":"/dataset/ai2d","name":"AI2D","full_name":"AI2 Diagrams","description_markdown":"AI2 Diagrams (AI2D) is a dataset of over 5000 grade school science diagrams with over 150000 rich annotations, their ground truth syntactic parses, and more than 15000 corresponding multiple choice questions.\r\n\r\nSource: [A Diagram Is Worth A Dozen Images](https://arxiv.org/pdf/1603.07396.pdf)\r\nImage Source: [Kembhavi et al](https://arxiv.org/pdf/1603.07396.pdf)","description_withheld":null,"homepage":"https://allenai.org/data/diagrams","introduced_date":"2016-03-24","introduced_date_note":null,"introduced_by":{"paper":"/paper/a-diagram-is-worth-a-dozen-images","title":"A Diagram Is Worth A Dozen Images","first_author":"Aniruddha Kembhavi","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Image Classification","url":"/task/image-classification","datasets_with_task":"/datasets/task/image-classification"},{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"}],"languages":[],"variants":["AI2D"],"data_loaders":[{"repo":"https://github.com/allenai/dqa-net","url":"https://github.com/allenai/dqa-net","frameworks":["tf"]}],"num_papers_in_archive":207,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-question-answering-vqa-on-ai2d","task":"Visual Question Answering (VQA)","dataset_variant":"AI2D","rows":4,"metrics":["EM"],"first_row_in_archive_order":{"model":"SMoLA-PaLI-X Specialist Model","paper":"/paper/omni-smola-boosting-generalist-multimodal","metrics":{"EM":"82.5"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/gemini-a-family-of-highly-capable-multimodal-1","title":"Gemini: A Family of Highly Capable Multimodal Models","date":"2023-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/omni-smola-boosting-generalist-multimodal","title":"Omni-SMoLA: Boosting Generalist Multimodal Models with Soft Mixture of Low-rank Experts","date":"2023-12-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/dublin-document-understanding-by-language","title":"DUBLIN -- Document Understanding By Language-Image Network","date":"2023-05-23","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}