{"url":"/dataset/math","name":"MATH","full_name":null,"description_markdown":"MATH is a new dataset of 12,500 challenging competition mathematics problems. Each problem in MATH has a full step-by-step solution which can be used to teach models to generate answer derivations and explanations.\r\n\r\nSource: [Hendrycks et al.](https://arxiv.org/pdf/2103.03874.pdf)\r\n\r\nImage source: [Hendrycks et al.](https://arxiv.org/pdf/2103.03874v1.pdf)","description_withheld":null,"homepage":"https://github.com/hendrycks/math/","introduced_date":"2021-03-05","introduced_date_note":null,"introduced_by":{"paper":"/paper/measuring-mathematical-problem-solving-with","title":"Measuring Mathematical Problem Solving With the MATH Dataset","first_author":"Dan Hendrycks","url":null},"license":{"name":"MIT","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Math Word Problem Solving","url":"/task/math-word-problem-solving","datasets_with_task":"/datasets/task/math-word-problem-solving"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["MATH","MATH minival"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/nivektk/math-augmented-dataset","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/hendrycks/math","url":"https://github.com/hendrycks/math","frameworks":["pytorch"]}],"num_papers_in_archive":1330,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/math-word-problem-solving-on-math","task":"Math Word Problem Solving","dataset_variant":"MATH","rows":135,"metrics":["Accuracy","Parameters (Billions)"],"first_row_in_archive_order":{"model":"Gemini 2.0 Flash Experimental","paper":null,"metrics":{"Accuracy":"89.7"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/math-word-problem-solving-on-math-minival","task":"Math Word Problem Solving","dataset_variant":"MATH minival","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Process Supervision (GPT-4)","paper":"/paper/let-s-verify-step-by-step-1","metrics":{"Accuracy":"78.2"},"code_links":[{"title":"openai/prm800k","url":"https://github.com/openai/prm800k"},{"title":"gentopia-ai/gentopia","url":"https://github.com/gentopia-ai/gentopia"},{"title":"consequentai/fneval","url":"https://github.com/consequentai/fneval"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/openmathinstruct-2-accelerating-ai-for-math","title":"OpenMathInstruct-2: Accelerating AI for Math with Massive Open-Source Instruction Data","date":"2024-10-02","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":0,"samples_unverified":15,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/qwen2-5-math-technical-report-toward","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","date":"2024-09-18","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/qwen2-technical-report","title":"Qwen2 Technical Report","date":"2024-07-15","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/step-dpo-step-wise-preference-optimization","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","date":"2024-06-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":7,"samples_unverified":5,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dart-math-difficulty-aware-rejection-tuning-1","title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","date":"2024-06-18","rows_on_this_dataset":8,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":10,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/alphamath-almost-zero-process-supervision","title":"AlphaMath Almost Zero: Process Supervision without Process","date":"2024-05-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/toward-self-improvement-of-llms-via","title":"Toward Self-Improvement of LLMs via Imagination, Searching, and Criticizing","date":"2024-04-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":13,"samples_unverified":5,"pointer_only_for_licence":18,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/macm-utilizing-a-multi-agent-system-for","title":"MACM: Utilizing a Multi-Agent System for Condition Mining in Solving Complex Mathematical Problems","date":"2024-04-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/branch-train-mix-mixing-expert-llms-into-a","title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","date":"2024-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/key-point-driven-data-synthesis-with-its","title":"Key-Point-Driven Data Synthesis with its Enhancement on Mathematical Reasoning","date":"2024-03-04","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-data-ability-boundary","title":"An Empirical Study of Data Ability Boundary in LLMs' Math Reasoning","date":"2024-02-23","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":9,"samples_unverified":2,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/openmathinstruct-1-a-1-8-million-math","title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","date":"2024-02-15","rows_on_this_dataset":12,"code_links":1,"syntology":null},{"paper":"/paper/deepseekmath-pushing-the-limits-of","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","date":"2024-02-05","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":8,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/augmenting-math-word-problems-via-iterative","title":"Augmenting Math Word Problems via Iterative Question Composing","date":"2024-01-17","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mixtral-of-experts","title":"Mixtral of Experts","date":"2024-01-08","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/parameter-efficient-sparsity-crafting-from","title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","date":"2024-01-05","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/gemini-a-family-of-highly-capable-multimodal-1","title":"Gemini: A Family of Highly Capable Multimodal Models","date":"2023-12-19","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/math-shepherd-a-label-free-step-by-step","title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","date":"2023-12-14","rows_on_this_dataset":3,"code_links":3,"syntology":null},{"paper":"/paper/mistral-7b","title":"Mistral 7B","date":"2023-10-10","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":9,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/query-and-response-augmentation-cannot-help","title":"MuggleMath: Assessing the Impact of Query and Response Augmentation on Math Reasoning","date":"2023-10-09","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/mathcoder-seamless-code-integration-in-llms","title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","date":"2023-10-05","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tora-a-tool-integrated-reasoning-agent-for","title":"ToRA: A Tool-Integrated Reasoning Agent for Mathematical Problem Solving","date":"2023-09-29","rows_on_this_dataset":8,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":5,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/metamath-bootstrap-your-own-mathematical","title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","date":"2023-09-21","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":22,"samples_ran":15,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/openchat-advancing-open-source-language","title":"OpenChat: Advancing Open-source Language Models with Mixed-Quality Data","date":"2023-09-20","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/wizardmath-empowering-mathematical-reasoning","title":"WizardMath: Empowering Mathematical Reasoning for Large Language Models via Reinforced Evol-Instruct","date":"2023-08-18","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":10,"samples_unverified":6,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/solving-challenging-math-word-problems-using","title":"Solving Challenging Math Word Problems Using GPT-4 Code Interpreter with Code-based Self-Verification","date":"2023-08-15","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/cumulative-reasoning-with-large-language","title":"Cumulative Reasoning with Large Language Models","date":"2023-08-08","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/skills-in-context-prompting-unlocking","title":"Skills-in-Context Prompting: Unlocking Compositionality in Large Language Models","date":"2023-08-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/let-s-verify-step-by-step-1","title":"Let's Verify Step by Step","date":"2023-05-31","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/palm-2-technical-report-1","title":"PaLM 2 Technical Report","date":"2023-05-17","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/progressive-hint-prompting-improves-reasoning","title":"Progressive-Hint Prompting Improves Reasoning in Large Language Models","date":"2023-04-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sparks-of-artificial-general-intelligence","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","date":"2023-03-22","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","rows_on_this_dataset":8,"code_links":57,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":58,"samples_ran":26,"samples_unverified":32,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pal-program-aided-language-models","title":"PAL: Program-aided Language Models","date":"2022-11-18","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","rows_on_this_dataset":7,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/solving-quantitative-reasoning-problems-with","title":"Solving Quantitative Reasoning Problems with Language Models","date":"2022-06-29","rows_on_this_dataset":13,"code_links":1,"syntology":null},{"paper":"/paper/measuring-mathematical-problem-solving-with","title":"Measuring Mathematical Problem Solving With the MATH Dataset","date":"2021-03-05","rows_on_this_dataset":8,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":23,"samples_harvested":255,"samples_ran":140,"samples_unverified":115,"pointer_only_for_licence":78,"papers_with_no_sample_that_ran":5,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}