{"url":"/dataset/econlogicqa","name":"EconLogicQA","full_name":null,"description_markdown":"EconLogicQA is a benchmark designed to test the sequential reasoning skills of large language models (LLMs) in economics, business, and supply chain management. It diverges from typical benchmarks by requiring models to understand and sequence multiple interconnected events, capturing complex economic logics. The benchmark includes multi-event scenarios and a thorough suite of evaluations to assess proficiency in economic contexts.","description_withheld":null,"homepage":"https://huggingface.co/datasets/yinzhu-quan/econ_logic_qa","introduced_date":"2024-05-13","introduced_date_note":null,"introduced_by":{"paper":"/paper/econlogicqa-a-question-answering-benchmark","title":"EconLogicQA: A Question-Answering Benchmark for Evaluating Large Language Models in Economic Sequential Reasoning","first_author":"Yinzhu Quan","url":null},"license":{"name":"CC BY-NC-SA 4.0","url":null},"modalities":[],"tasks":[{"name":"Sentence Ordering","url":"/task/sentence-ordering","datasets_with_task":"/datasets/task/sentence-ordering"}],"languages":[],"variants":["EconLogicQA"],"data_loaders":[],"num_papers_in_archive":2,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/sentence-ordering-on-econlogicqa","task":"Sentence Ordering","dataset_variant":"EconLogicQA","rows":18,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GPT-4-Turbo","paper":"/paper/econlogicqa-a-question-answering-benchmark","metrics":{"Accuracy":"0.5692"},"code_links":[{"title":"yinzhu-quan/lm-evaluation-harness","url":"https://github.com/yinzhu-quan/lm-evaluation-harness"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/econlogicqa-a-question-answering-benchmark","title":"EconLogicQA: A Question-Answering Benchmark for Evaluating Large Language Models in Economic Sequential Reasoning","date":"2024-05-13","rows_on_this_dataset":18,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}