{"url":"/dataset/arc","name":"ARC (AI2 Reasoning Challenge)","full_name":null,"description_markdown":"The AI2’s Reasoning Challenge (**ARC**) dataset is a multiple-choice question-answering dataset, containing questions from science exams from grade 3 to grade 9. The dataset is split in two partitions: Easy and Challenge, where the latter partition contains the more difficult questions that require reasoning. Most of the questions have 4 answer choices, with <1% of all the questions having either 3 or 5 answer choices. ARC includes a supporting KB of 14.3M unstructured text passages.\r\n\r\nSource: [Quick and (not so) Dirty: Unsupervised Selection of Justification Sentences for Multi-hop Question Answering](https://arxiv.org/abs/1911.07176)\r\nImage Source: [https://arxiv.org/abs/1803.05457](https://arxiv.org/abs/1803.05457)","description_withheld":null,"homepage":"https://allenai.org/data/arc","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/think-you-have-solved-question-answering-try","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","first_author":"Peter Clark","url":null},"license":{"name":"CC BY-SA 4.0","url":"https://allenai.org/data/arc"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Common Sense Reasoning","url":"/task/common-sense-reasoning","datasets_with_task":"/datasets/task/common-sense-reasoning"},{"name":"Stance Detection","url":"/task/stance-detection","datasets_with_task":"/datasets/task/stance-detection"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["ARC (Easy)","ARC (Challenge)","ARC (AI2 Reasoning Challenge)"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/malhajar/arc-tr","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/ai2_arc_with_ir","frameworks":["tf","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/ai2_arc","frameworks":["tf","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/arc","frameworks":["tf","jax"]}],"num_papers_in_archive":178,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/common-sense-reasoning-on-arc-challenge","task":"Common Sense Reasoning","dataset_variant":"ARC (Challenge)","rows":54,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GPT-4 (few-shot, k=25)","paper":"/paper/gpt-4-technical-report-1","metrics":{"Accuracy":"96.4"},"code_links":[{"title":"openai/evals","url":"https://github.com/openai/evals"},{"title":"shmsw25/factscore","url":"https://github.com/shmsw25/factscore"},{"title":"unispac/visual-adversarial-examples-jailbreak-large-language-models","url":"https://github.com/unispac/visual-adversarial-examples-jailbreak-large-language-models"},{"title":"gpt4life/alpagasus","url":"https://github.com/gpt4life/alpagasus"},{"title":"emrgnt-cmplxty/zero-shot-replication","url":"https://github.com/emrgnt-cmplxty/zero-shot-replication"},{"title":"ethz-privsec/superhuman-ai-consistency","url":"https://github.com/ethz-privsec/superhuman-ai-consistency"},{"title":"ethz-spylab/superhuman-ai-consistency","url":"https://github.com/ethz-spylab/superhuman-ai-consistency"},{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"},{"title":"AUCOHL/RTL-Repo","url":"https://github.com/AUCOHL/RTL-Repo"},{"title":"zach-zhiling-zheng/reticular_chemist","url":"https://github.com/zach-zhiling-zheng/reticular_chemist"},{"title":"lflage/openfactscore","url":"https://github.com/lflage/openfactscore"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/common-sense-reasoning-on-arc-easy","task":"Common Sense Reasoning","dataset_variant":"ARC (Easy)","rows":47,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"ST-MoE-32B 269B (fine-tuned)","paper":"/paper/designing-effective-sparse-expert-models","metrics":{"Accuracy":"95.2"},"code_links":[{"title":"tensorflow/mesh","url":"https://github.com/tensorflow/mesh"},{"title":"xuefuzhao/openmoe","url":"https://github.com/xuefuzhao/openmoe"},{"title":"yikangshen/megablocks","url":"https://github.com/yikangshen/megablocks"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/stance-detection-on-arc","task":"Stance Detection","dataset_variant":"ARC (AI2 Reasoning Challenge)","rows":1,"metrics":["F1"],"first_row_in_archive_order":{"model":"TESTED","paper":"/paper/topic-guided-sampling-for-data-efficient","metrics":{"F1":"64.82"},"code_links":[{"title":"copenlu/TESTED","url":"https://github.com/copenlu/TESTED"},{"title":"copenlu/TESTED","url":"https://github.com/copenlu/TESTED/blob/main/README.md"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/mixture-of-subspaces-in-low-rank-adaptation","title":"Mixture-of-Subspaces in Low-Rank Adaptation","date":"2024-06-16","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mixlora-enhancing-large-language-models-fine","title":"MixLoRA: Enhancing Large Language Models Fine-Tuning with LoRA-based Mixture of Experts","date":"2024-04-22","rows_on_this_dataset":6,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mixtral-of-experts","title":"Mixtral of Experts","date":"2024-01-08","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/parameter-efficient-sparsity-crafting-from","title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","date":"2024-01-05","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/mamba-linear-time-sequence-modeling-with","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","date":"2023-12-01","rows_on_this_dataset":1,"code_links":35,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":62,"samples_ran":18,"samples_unverified":44,"pointer_only_for_licence":28,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mistral-7b","title":"Mistral 7B","date":"2023-10-10","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":9,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/textbooks-are-all-you-need-ii-phi-1-5","title":"Textbooks Are All You Need II: phi-1.5 technical report","date":"2023-09-11","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/model-card-and-evaluations-for-claude-models","title":"Model Card and Evaluations for Claude Models","date":"2023-07-11","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/stay-on-topic-with-classifier-free-guidance","title":"Stay on topic with Classifier-Free Guidance","date":"2023-06-30","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/topic-guided-sampling-for-data-efficient","title":"Topic-Guided Sampling For Data-Efficient Multi-Domain Stance Detection","date":"2023-06-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/palm-2-technical-report-1","title":"PaLM 2 Technical Report","date":"2023-05-17","rows_on_this_dataset":7,"code_links":1,"syntology":null},{"paper":"/paper/pythia-a-suite-for-analyzing-large-language","title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","date":"2023-04-03","rows_on_this_dataset":4,"code_links":4,"syntology":null},{"paper":"/paper/bloomberggpt-a-large-language-model-for","title":"BloombergGPT: A Large Language Model for Finance","date":"2023-03-30","rows_on_this_dataset":8,"code_links":2,"syntology":null},{"paper":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","rows_on_this_dataset":2,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","rows_on_this_dataset":8,"code_links":57,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":58,"samples_ran":26,"samples_unverified":32,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/massive-language-models-can-be-accurately","title":"SparseGPT: Massive Language Models Can Be Accurately Pruned in One-Shot","date":"2023-01-02","rows_on_this_dataset":10,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":2,"samples_unverified":10,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","rows_on_this_dataset":8,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/large-language-models-can-self-improve","title":"Large Language Models Can Self-Improve","date":"2022-10-20","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/unifying-language-learning-paradigms","title":"UL2: Unifying Language Learning Paradigms","date":"2022-05-10","rows_on_this_dataset":6,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":0,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/designing-effective-sparse-expert-models","title":"ST-MoE: Designing Stable and Transferable Sparse Expert Models","date":"2022-02-17","rows_on_this_dataset":4,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/glam-efficient-scaling-of-language-models","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","date":"2021-12-13","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/finetuned-language-models-are-zero-shot","title":"Finetuned Language Models Are Zero-Shot Learners","date":"2021-09-03","rows_on_this_dataset":4,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","rows_on_this_dataset":4,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":65,"samples_ran":15,"samples_unverified":50,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":13,"samples_harvested":259,"samples_ran":92,"samples_unverified":167,"pointer_only_for_licence":58,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}