{"url":"/task/elementary-mathematics","name":"Elementary Mathematics","slug":"elementary-mathematics","description_markdown":null,"categories":[{"name":"Methodology","url":"/area/methodology"},{"name":"Miscellaneous","url":"/area/miscellaneous"},{"name":"Reasoning","url":"/area/reasoning"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":7,"papers_with_code":5,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":5,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/elementary-mathematics-on-big-bench","slug":"elementary-mathematics-on-big-bench","dataset":"BIG-bench","dataset_url":"/dataset/big-bench","rows_in_archive":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper_title":"Scaling Language Models: Methods, Analysis & Insights from Training Gopher","paper_url":"/paper/scaling-language-models-methods-analysis-1","paper_date":"2021-12-08","arxiv_id":"2112.11446","code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}],"syntology":null}}],"datasets":[{"url":"/dataset/big-bench","name":"BIG-bench","full_name":"Beyond the Imitation Game Benchmark","num_papers_in_archive":349},{"url":"/dataset/mathbench","name":"MathBench","full_name":"MathBench: Evaluating the Theory and Application Proficiency of LLMs with a Hierarchical Mathematics Benchmark","num_papers_in_archive":16},{"url":"/dataset/dart-math-hard","name":"DART-Math-Hard","full_name":"","num_papers_in_archive":2},{"url":"/dataset/dart-math-uniform","name":"DART-Math-Uniform","full_name":"","num_papers_in_archive":1},{"url":"/dataset/ncte-transcripts","name":"NCTE Transcripts","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/logical-reasoning","name":"Logical Reasoning"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":5,"of":5,"tagged_in_all":7,"items":[{"url":"/paper/measuring-massive-multitask-language","title":"Measuring Massive Multitask Language Understanding","date":"2020-09-07","arxiv_id":"2009.03300","repositories_listed":18,"syntology":{"n":26,"n_ran":14,"n_unverified":12,"n_pointer_only":1}},{"url":"/paper/scaling-language-models-methods-analysis-1","title":"Scaling Language Models: Methods, Analysis & Insights from Training Gopher","date":"2021-12-08","arxiv_id":"2112.11446","repositories_listed":3,"syntology":null},{"url":"/paper/an-empirical-study-on-challenging-math","title":"MathChat: Converse to Tackle Challenging Math Problems with LLM Agents","date":"2023-06-02","arxiv_id":"2306.01337","repositories_listed":2,"syntology":null},{"url":"/paper/mathematical-capabilities-of-chatgpt-1","title":"Mathematical Capabilities of ChatGPT","date":"2023-01-31","arxiv_id":"2301.13867","repositories_listed":2,"syntology":null},{"url":"/paper/the-ncte-transcripts-a-dataset-of-elementary","title":"The NCTE Transcripts: A Dataset of Elementary Math Classroom Transcripts","date":"2022-11-21","arxiv_id":"2211.11772","repositories_listed":1,"syntology":null}],"syntology_records":1,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-25T09:33:49+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}