{"url":"/dataset/openbookqa","name":"OpenBookQA","full_name":"OBQA","description_markdown":"**OpenBookQA** is a new kind of question-answering dataset modeled after open book exams for assessing human understanding of a subject. It consists of 5,957 multiple-choice elementary-level science questions (4,957 train, 500 dev, 500 test), which probe the understanding of a small “book” of 1,326 core science facts and the application of these facts to novel situations. For training, the dataset includes a mapping from each question to the core science fact it was designed to probe. Answering OpenBookQA questions requires additional broad common knowledge, not contained in the book. The questions, by design, are answered incorrectly by both a retrieval-based algorithm and a word co-occurrence algorithm.\r\nAdditionally, the dataset includes a collection of 5,167 crowd-sourced common knowledge facts, and an expanded version of the train/dev/test questions where each question is associated with its originating core fact, a human accuracy score, a clarity score, and an anonymized crowd-worker ID.\r\n\r\nSource: [https://allenai.org/data/open-book-qa](https://allenai.org/data/open-book-qa)\r\nImage Source: [https://arxiv.org/pdf/1809.02789.pdf](https://arxiv.org/pdf/1809.02789.pdf)","description_withheld":null,"homepage":"https://allenai.org/data/open-book-qa","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/can-a-suit-of-armor-conduct-electricity-a-new","title":"Can a Suit of Armor Conduct Electricity? A New Dataset for Open Book Question Answering","first_author":"Todor Mihaylov","url":null},"license":{"name":"Custom","url":"https://github.com/allenai/OpenBookQA"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["OpenBookQA","OBQA"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/openbookqa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/openbookqa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/bobox/OpenbookQA-4ST","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/attacker-exploiting-everyone/test-dataset","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/bobox/openbookqa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/Kaggle/kaggle-api","url":"https://www.kaggle.com/datasets/konradb/chatbot-dataset-openbookqa","frameworks":[]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/openbookqa","frameworks":["tf","jax"]}],"num_papers_in_archive":635,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-openbookqa","task":"Question Answering","dataset_variant":"OpenBookQA","rows":45,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GPT-4 + knowledge base","paper":null,"metrics":{"Accuracy":"95.9"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/question-answering-on-obqa","task":"Question Answering","dataset_variant":"OBQA","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"FLAN 137B (zero-shot)","paper":"/paper/finetuned-language-models-are-zero-shot","metrics":{"Accuracy":"78.4"},"code_links":[{"title":"hiyouga/llama-efficient-tuning","url":"https://github.com/hiyouga/llama-efficient-tuning"},{"title":"bigcode-project/starcoder","url":"https://github.com/bigcode-project/starcoder"},{"title":"bigscience-workshop/promptsource","url":"https://github.com/bigscience-workshop/promptsource"},{"title":"google-research/flan","url":"https://github.com/google-research/flan"},{"title":"ukplab/arxiv2025-inherent-limits-plms","url":"https://github.com/ukplab/arxiv2025-inherent-limits-plms"},{"title":"openbiolink/promptsource","url":"https://github.com/openbiolink/promptsource"},{"title":"MS-P3/code6","url":"https://github.com/MS-P3/code6/tree/main/finetune"},{"title":"hojjat-mokhtarabadi/promptsource","url":"https://github.com/hojjat-mokhtarabadi/promptsource"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-openbookqa","task":"Text Generation","dataset_variant":"OpenBookQA","rows":0,"metrics":["acc"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/mixture-of-subspaces-in-low-rank-adaptation","title":"Mixture-of-Subspaces in Low-Rank Adaptation","date":"2024-06-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mixlora-enhancing-large-language-models-fine","title":"MixLoRA: Enhancing Large Language Models Fine-Tuning with LoRA-based Mixture of Experts","date":"2024-04-22","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/palm-2-technical-report-1","title":"PaLM 2 Technical Report","date":"2023-05-17","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/lamini-lm-a-diverse-herd-of-distilled-models","title":"LaMini-LM: A Diverse Herd of Distilled Models from Large-Scale Instructions","date":"2023-04-27","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/bloomberggpt-a-large-language-model-for","title":"BloombergGPT: A Large Language Model for Finance","date":"2023-03-30","rows_on_this_dataset":4,"code_links":2,"syntology":null},{"paper":"/paper/grapeqa-graph-augmentation-and-pruning-to","title":"GrapeQA: GRaph Augmentation and Pruning to Enhance Question-Answering","date":"2023-03-22","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","rows_on_this_dataset":4,"code_links":57,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":58,"samples_ran":26,"samples_unverified":32,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/large-language-models-can-self-improve","title":"Large Language Models Can Self-Improve","date":"2022-10-20","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/clues-before-answers-generation-enhanced","title":"Clues Before Answers: Generation-Enhanced Multiple-Choice QA","date":"2022-04-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","rows_on_this_dataset":2,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":37,"samples_ran":30,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gnn-is-a-counter-revisiting-gnn-for-question","title":"GNN is a Counter? Revisiting GNN for Question Answering","date":"2021-10-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/finetuned-language-models-are-zero-shot","title":"Finetuned Language Models Are Zero-Shot Learners","date":"2021-09-03","rows_on_this_dataset":2,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/qa-gnn-reasoning-with-language-models-and","title":"QA-GNN: Reasoning with Language Models and Knowledge Graphs for Question Answering","date":"2021-04-13","rows_on_this_dataset":3,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":1,"samples_unverified":24,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fusing-context-into-knowledge-graph-for","title":"Fusing Context Into Knowledge Graph for Commonsense Question Answering","date":"2020-12-09","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","rows_on_this_dataset":2,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":65,"samples_ran":15,"samples_unverified":50,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unifiedqa-crossing-format-boundaries-with-a","title":"UnifiedQA: Crossing Format Boundaries With a Single QA System","date":"2020-05-02","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/careful-selection-of-knowledge-to-solve-open","title":"Careful Selection of Knowledge to solve Open Book Question Answering","date":"2019-07-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/can-a-suit-of-armor-conduct-electricity-a-new","title":"Can a Suit of Armor Conduct Electricity? A New Dataset for Open Book Question Answering","date":"2018-09-08","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":10,"samples_harvested":224,"samples_ran":91,"samples_unverified":133,"pointer_only_for_licence":19,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}