{"url":"/dataset/a-okvqa","name":"A-OKVQA","full_name":null,"description_markdown":"**A-OKVQA** is crowdsourced visual question answering dataset composed of a diverse set of about 25K questions requiring a broad base of commonsense and world knowledge to answer.","description_withheld":null,"homepage":"https://allenai.org/project/a-okvqa/home","introduced_date":"2022-06-03","introduced_date_note":null,"introduced_by":{"paper":"/paper/a-okvqa-a-benchmark-for-visual-question","title":"A-OKVQA: A Benchmark for Visual Question Answering using World Knowledge","first_author":"Dustin Schwenk","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"}],"languages":[],"variants":["A-OKVQA"],"data_loaders":[],"num_papers_in_archive":154,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-question-answering-on-a-okvqa","task":"Visual Question Answering (VQA)","dataset_variant":"A-OKVQA","rows":15,"metrics":["MC Accuracy","DA VQA Score"],"first_row_in_archive_order":{"model":"SMoLA-PaLI-X Specialist Model","paper":"/paper/omni-smola-boosting-generalist-multimodal","metrics":{"DA VQA Score":"70.55","MC Accuracy":"83.75"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/hydra-a-hyper-agent-for-dynamic-compositional","title":"HYDRA: A Hyper Agent for Dynamic Compositional Visual Reasoning","date":"2024-03-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/visual-program-distillation-distilling-tools","title":"Visual Program Distillation: Distilling Tools and Programmatic Reasoning into Vision-Language Models","date":"2023-12-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/omni-smola-boosting-generalist-multimodal","title":"Omni-SMoLA: Boosting Generalist Multimodal Models with Soft Mixture of Low-rank Experts","date":"2023-12-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/boosting-the-power-of-small-multimodal","title":"Boosting the Power of Small Multimodal Reasoning Models to Match Larger Models with Self-Consistency Training","date":"2023-11-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-simple-baseline-for-knowledge-based-visual","title":"A Simple Baseline for Knowledge-Based Visual Question Answering","date":"2023-10-20","rows_on_this_dataset":1,"code_links":0,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/prompting-large-language-models-with-answer","title":"Prophet: Prompting Large Language Models with Complementary Answer Heuristics for Knowledge-based Visual Question Answering","date":"2023-03-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/promptcap-prompt-guided-task-aware-image","title":"PromptCap: Prompt-Guided Task-Aware Image Captioning","date":"2022-11-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vlc-bert-visual-question-answering-with","title":"VLC-BERT: Visual Question Answering with Contextualized Commonsense Knowledge","date":"2022-10-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/webly-supervised-concept-expansion-for","title":"Webly Supervised Concept Expansion for General Purpose Vision Models","date":"2022-02-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/krisp-integrating-implicit-and-symbolic","title":"KRISP: Integrating Implicit and Symbolic Knowledge for Open-Domain Knowledge-Based VQA","date":"2020-12-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/lxmert-learning-cross-modality-encoder","title":"LXMERT: Learning Cross-Modality Encoder Representations from Transformers","date":"2019-08-20","rows_on_this_dataset":1,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":4,"samples_unverified":11,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vilbert-pretraining-task-agnostic","title":"ViLBERT: Pretraining Task-Agnostic Visiolinguistic Representations for Vision-and-Language Tasks","date":"2019-08-06","rows_on_this_dataset":3,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":34,"samples_ran":10,"samples_unverified":24,"pointer_only_for_licence":34,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pythia-v01-the-winning-entry-to-the-vqa","title":"Pythia v0.1: the Winning Entry to the VQA Challenge 2018","date":"2018-07-26","rows_on_this_dataset":1,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":8,"samples_harvested":73,"samples_ran":24,"samples_unverified":49,"pointer_only_for_licence":43,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}