{"url":"/dataset/lila","name":"Lila","full_name":null,"description_markdown":"**Lila** is a unified mathematical reasoning benchmark consisting of 23 diverse tasks along four dimensions: (i) mathematical abilities e.g., arithmetic, calculus (ii) language format e.g., question-answering, fill-in-the-blanks (iii) language diversity e.g., no language, simple language (iv) external knowledge e.g., commonsense, physics. The benchmark is constructed by extending 20 datasets benchmark by collecting task instructions and solutions in the form of Python programs, thereby obtaining explainable solutions in addition to the correct answer.","description_withheld":null,"homepage":"https://github.com/allenai/Lila","introduced_date":"2022-10-31","introduced_date_note":null,"introduced_by":{"paper":"/paper/lila-a-unified-benchmark-for-mathematical","title":"Lila: A Unified Benchmark for Mathematical Reasoning","first_author":"Swaroop Mishra","url":null},"license":{"name":"CC-BY-4.0","url":"https://github.com/allenai/Lila/blob/main/LICENSE.txt"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Mathematical Reasoning","url":"/task/mathematical-reasoning","datasets_with_task":"/datasets/task/mathematical-reasoning"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Lila"],"data_loaders":[],"num_papers_in_archive":12,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}