{"url":"/dataset/svamp","name":"SVAMP","full_name":"Simple Variations on Arithmetic Math word Problems","description_markdown":"A challenge set for elementary-level Math Word Problems (MWP). An MWP consists of a short Natural Language narrative that describes a state of the world and poses a question about some unknown quantities.\r\n\r\nThe examples in **SVAMP** test a model across different aspects of solving MWPs: 1) Is the model question sensitive? 2) Does the model have robust reasoning ability? 3) Is it invariant to structural alterations?","description_withheld":null,"homepage":"https://github.com/arkilpatel/SVAMP","introduced_date":"2021-03-12","introduced_date_note":null,"introduced_by":{"paper":"/paper/are-nlp-models-really-able-to-solve-simple","title":"Are NLP Models really able to Solve Simple Math Word Problems?","first_author":"Arkil Patel","url":null},"license":{"name":"MIT","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Mathematical Reasoning","url":"/task/mathematical-reasoning","datasets_with_task":"/datasets/task/mathematical-reasoning"},{"name":"Math Word Problem Solving","url":"/task/math-word-problem-solving","datasets_with_task":"/datasets/task/math-word-problem-solving"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["SVAMP","SVAMP (1:N)"],"data_loaders":[{"repo":"https://github.com/arkilpatel/SVAMP","url":"https://github.com/arkilpatel/SVAMP","frameworks":["pytorch"]}],"num_papers_in_archive":362,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/math-word-problem-solving-on-svamp","task":"Math Word Problem Solving","dataset_variant":"SVAMP","rows":26,"metrics":["Execution Accuracy","Accuracy"],"first_row_in_archive_order":{"model":"GPT-4 (Teaching-Inspired)","paper":"/paper/teaching-inspired-integrated-prompting","metrics":{"Execution Accuracy":"93.9"},"code_links":[{"title":"sallytan13/teaching-inspired-prompting","url":"https://github.com/sallytan13/teaching-inspired-prompting"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/math-word-problem-solving-on-svamp-1-n","task":"Math Word Problem Solving","dataset_variant":"SVAMP (1:N)","rows":2,"metrics":["Execution Accuracy"],"first_row_in_archive_order":{"model":"ATHENA (roberta-large)","paper":"/paper/athena-mathematical-reasoning-with-thought","metrics":{"Execution Accuracy":"67.8"},"code_links":[{"title":"the-jb/athena-math","url":"https://github.com/the-jb/athena-math"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/teaching-inspired-integrated-prompting","title":"Teaching-Inspired Integrated Prompting Framework: A Novel Approach for Enhancing Reasoning in Large Language Models","date":"2024-10-10","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":8,"samples_ran":8,"samples_unverified":0,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/achieving-97-on-gsm8k-deeply-understanding","title":"Achieving >97% on GSM8K: Deeply Understanding the Problems Makes LLMs Better Solvers for Math Word Problems","date":"2024-04-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/an-empirical-study-of-data-ability-boundary","title":"An Empirical Study of Data Ability Boundary in LLMs' Math Reasoning","date":"2024-02-23","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":11,"samples_ran":9,"samples_unverified":2,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/openmathinstruct-1-a-1-8-million-math","title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","date":"2024-02-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/frugal-lms-trained-to-invoke-symbolic-solvers","title":"Frugal LMs Trained to Invoke Symbolic Solvers Achieve Parameter-Efficient Arithmetic Reasoning","date":"2023-12-09","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/athena-mathematical-reasoning-with-thought","title":"ATHENA: Mathematical Reasoning with Thought Expansion","date":"2023-11-02","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/mathcoder-seamless-code-integration-in-llms","title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","date":"2023-10-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llama-2-open-foundation-and-fine-tuned-chat","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","date":"2023-07-18","rows_on_this_dataset":1,"code_links":19,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":52,"samples_ran":31,"samples_unverified":21,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/math-word-problem-solving-by-generating","title":"Math Word Problem Solving by Generating Linguistic Variants of Problem Statements","date":"2023-06-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/does-chatgpt-comprehend-the-place-value-in","title":"Does ChatGPT Comprehend the Place Value in Numbers When Solving Math Word Problems?","date":"2023-06-03","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/learning-multi-step-reasoning-from-arithmetic","title":"Learning Multi-Step Reasoning by Solving Arithmetic Tasks","date":"2023-06-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/automatic-model-selection-with-large-language","title":"Automatic Model Selection with Large Language Models for Reasoning","date":"2023-05-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/progressive-hint-prompting-improves-reasoning","title":"Progressive-Hint Prompting Improves Reasoning in Large Language Models","date":"2023-04-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/large-language-models-are-zero-shot-reasoners","title":"Large Language Models are Zero-Shot Reasoners","date":"2022-05-24","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-to-reason-deductively-math-word","title":"Learning to Reason Deductively: Math Word Problem Solving as Complex Relation Extraction","date":"2022-03-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/are-nlp-models-really-able-to-solve-simple","title":"Are NLP Models really able to Solve Simple Math Word Problems?","date":"2021-03-12","rows_on_this_dataset":4,"code_links":3,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":9,"samples_harvested":87,"samples_ran":61,"samples_unverified":26,"pointer_only_for_licence":46,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}