{"url":"/task/gsm8k","name":"GSM8K","slug":"gsm8k","description_markdown":null,"categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":439,"papers_with_code":209,"benchmarks":1,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/gsm8k-on-gsm8k","slug":"gsm8k-on-gsm8k","dataset":"GSM8K","dataset_url":"/dataset/gsm8k","rows_in_archive":3,"metrics":["Accuracy","0-shot MRR"],"first_row_in_archive_order":{"model":"Xolver","paper_title":"Xolver: Multi-Agent Reasoning with Holistic Experience Learning Just Like an Olympiad Team","paper_url":"/paper/xolver-multi-agent-reasoning-with-holistic","paper_date":"2025-06-17","arxiv_id":"2506.14234","code_links":[{"title":"kagnlp/Xolver","url":"https://github.com/kagnlp/Xolver"}],"syntology":null}},{"leaderboard":null,"slug":"gsm8k-on-gsm8k-5-shots","dataset":"gsm8k (5-shots)","dataset_url":"/dataset/gsm8k","rows_in_archive":0,"metrics":["exact_match_strict"],"first_row_in_archive_order":null}],"datasets":[{"url":"/dataset/gsm8k","name":"GSM8K","full_name":"","num_papers_in_archive":1881}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":209,"tagged_in_all":439,"items":[{"url":"/paper/chain-of-thought-prompting-elicits-reasoning","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","date":"2022-01-28","arxiv_id":"2201.11903","repositories_listed":19,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/chatglm-a-family-of-large-language-models","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","date":"2024-06-18","arxiv_id":"2406.12793","repositories_listed":7,"syntology":{"n":29,"n_ran":15,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/qwen2-technical-report","title":"Qwen2 Technical Report","date":"2024-07-15","arxiv_id":"2407.10671","repositories_listed":6,"syntology":null},{"url":"/paper/training-verifiers-to-solve-math-word","title":"Training Verifiers to Solve Math Word Problems","date":"2021-10-27","arxiv_id":"2110.14168","repositories_listed":6,"syntology":{"n":7,"n_ran":1,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/large-language-models-as-optimizers","title":"Large Language Models as Optimizers","date":"2023-09-07","arxiv_id":"2309.03409","repositories_listed":4,"syntology":{"n":14,"n_ran":12,"n_unverified":2,"n_pointer_only":2}},{"url":"/paper/language-models-are-multilingual-chain-of","title":"Language Models are Multilingual Chain-of-Thought Reasoners","date":"2022-10-06","arxiv_id":"2210.03057","repositories_listed":4,"syntology":null},{"url":"/paper/large-language-models-are-zero-shot-reasoners","title":"Large Language Models are Zero-Shot Reasoners","date":"2022-05-24","arxiv_id":"2205.11916","repositories_listed":4,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/bone-block-affine-transformation-as-parameter","title":"Balancing LoRA Performance and Efficiency with Simple Shard Sharing","date":"2024-09-19","arxiv_id":"2409.15371","repositories_listed":3,"syntology":null},{"url":"/paper/mutual-reasoning-makes-smaller-llms-stronger","title":"Mutual Reasoning Makes Smaller LLMs Stronger Problem-Solvers","date":"2024-08-12","arxiv_id":"2408.06195","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/math-shepherd-a-label-free-step-by-step","title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","date":"2023-12-14","arxiv_id":"2312.08935","repositories_listed":3,"syntology":null},{"url":"/paper/language-models-are-super-mario-absorbing","title":"Language Models are Super Mario: Absorbing Abilities from Homologous Models as a Free Lunch","date":"2023-11-06","arxiv_id":"2311.03099","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/askit-unified-programming-interface-for","title":"AskIt: Unified Programming Interface for Programming with Large Language Models","date":"2023-08-29","arxiv_id":"2308.15645","repositories_listed":3,"syntology":null},{"url":"/paper/kernel-ssl-kernel-kl-divergence-for-self","title":"Matrix Information Theory for Self-Supervised Learning","date":"2023-05-27","arxiv_id":"2305.17326","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":5}},{"url":"/paper/pal-program-aided-language-models","title":"PAL: Program-aided Language Models","date":"2022-11-18","arxiv_id":"2211.10435","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/self-consistency-improves-chain-of-thought","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","date":"2022-03-21","arxiv_id":"2203.11171","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/let-llms-break-free-from-overthinking-via","title":"Let LLMs Break Free from Overthinking via Self-Braking Tuning","date":"2025-05-20","arxiv_id":"2505.14604","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/offline-reinforcement-learning-for-llm-multi","title":"Offline Reinforcement Learning for LLM Multi-Step Reasoning","date":"2024-12-20","arxiv_id":"2412.16145","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/supercorrect-supervising-and-correcting","title":"SuperCorrect: Supervising and Correcting Language Models with Error-Driven Insights","date":"2024-10-11","arxiv_id":"2410.09008","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/enhancing-multi-step-reasoning-abilities-of","title":"Enhancing Multi-Step Reasoning Abilities of Language Models through Direct Q-Function Optimization","date":"2024-10-11","arxiv_id":"2410.09302","repositories_listed":2,"syntology":null},{"url":"/paper/omni-math-a-universal-olympiad-level","title":"Omni-MATH: A Universal Olympiad Level Mathematic Benchmark For Large Language Models","date":"2024-10-10","arxiv_id":"2410.07985","repositories_listed":2,"syntology":{"n":19,"n_ran":4,"n_unverified":15,"n_pointer_only":19}},{"url":"/paper/gsm-symbolic-understanding-the-limitations-of","title":"GSM-Symbolic: Understanding the Limitations of Mathematical Reasoning in Large Language Models","date":"2024-10-07","arxiv_id":"2410.05229","repositories_listed":2,"syntology":null},{"url":"/paper/large-language-monkeys-scaling-inference","title":"Large Language Monkeys: Scaling Inference Compute with Repeated Sampling","date":"2024-07-31","arxiv_id":"2407.21787","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":5}},{"url":"/paper/monte-carlo-tree-search-boosts-reasoning-via","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","date":"2024-05-01","arxiv_id":"2405.00451","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/quiet-star-language-models-can-teach","title":"Quiet-STaR: Language Models Can Teach Themselves to Think Before Speaking","date":"2024-03-14","arxiv_id":"2403.09629","repositories_listed":2,"syntology":{"n":9,"n_ran":4,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/common-7b-language-models-already-possess","title":"Common 7B Language Models Already Possess Strong Math Capabilities","date":"2024-03-07","arxiv_id":"2403.04706","repositories_listed":2,"syntology":{"n":28,"n_ran":6,"n_unverified":22,"n_pointer_only":12}},{"url":"/paper/automathtext-autonomous-data-selection-with","title":"Autonomous Data Selection with Zero-shot Generative Classifiers for Mathematical Texts","date":"2024-02-12","arxiv_id":"2402.07625","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/mario-math-reasoning-with-code-interpreter","title":"MARIO: MAth Reasoning with code Interpreter Output -- A Reproducible Pipeline","date":"2024-01-16","arxiv_id":"2401.08190","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/challenge-llms-to-reason-about-reasoning-a","title":"MR-GSM8K: A Meta-Reasoning Benchmark for Large Language Model Evaluation","date":"2023-12-28","arxiv_id":"2312.17080","repositories_listed":2,"syntology":null},{"url":"/paper/data-contamination-quiz-a-tool-to-detect-and","title":"Data Contamination Quiz: A Tool to Detect and Estimate Contamination in Large Language Models","date":"2023-11-10","arxiv_id":"2311.06233","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/breaking-language-barriers-in-multilingual","title":"Breaking Language Barriers in Multilingual Mathematical Reasoning: Insights and Observations","date":"2023-10-31","arxiv_id":"2310.20246","repositories_listed":2,"syntology":{"n":11,"n_ran":7,"n_unverified":4,"n_pointer_only":0}}],"syntology_records":22,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}