{"url":"/dataset/arcbench","name":"ArcBench","full_name":null,"description_markdown":"ArcBench is a logically challenging dataset of 158 English question–answer pairs, derived from the RoR-Bench benchmark. It targets deductive and multi-step reasoning in LLMs and multi-agent systems. The dataset was curated by translating original riddles into accessible English, removing multi-modal complexities, and validating each pair through automated reasoning workflows (e.g., Nexus Architect). ArcBench supports evaluation of reasoning performance, workflow refinement with feedback loops, comparative analysis of language models, and prompt-engineering research.","description_withheld":null,"homepage":"https://github.com/PrimisAI/arcbench","introduced_date":"2025-07-18","introduced_date_note":null,"introduced_by":{"paper":"/paper/adaptive-multi-agent-reasoning-via-automated","title":"Adaptive Multi-Agent Reasoning via Automated Workflow Generation","first_author":"Humza Sami","url":null},"license":{"name":"Apache 2.0","url":"https://github.com/PrimisAI/arcbench/blob/main/LICENSE"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["ArcBench"],"data_loaders":[],"num_papers_in_archive":0,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}