{"url":"/dataset/gsm8k","name":"GSM8K","full_name":null,"description_markdown":"GSM8K is a dataset of 8.5K high quality linguistically diverse grade school math word problems created by human problem writers. The dataset is segmented into 7.5K training problems and 1K test problems. These problems take between 2 and 8 steps to solve, and solutions primarily involve performing a sequence of elementary calculations using basic arithmetic operations (+ − ×÷) to reach the final answer. A bright middle school student should be able to solve every problem. It can be used for multi-step mathematical reasoning.\r\n\r\nImage source: [https://arxiv.org/pdf/2110.14168v1.pdf](https://arxiv.org/pdf/2110.14168v1.pdf)","description_withheld":null,"homepage":"https://github.com/openai/grade-school-math","introduced_date":"2021-10-27","introduced_date_note":null,"introduced_by":{"paper":"/paper/training-verifiers-to-solve-math-word","title":"Training Verifiers to Solve Math Word Problems","first_author":"Karl Cobbe","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Mathematical Reasoning","url":"/task/mathematical-reasoning","datasets_with_task":"/datasets/task/mathematical-reasoning"},{"name":"Arithmetic Reasoning","url":"/task/arithmetic-reasoning","datasets_with_task":"/datasets/task/arithmetic-reasoning"},{"name":"GSM8K","url":"/task/gsm8k","datasets_with_task":"/datasets/task/gsm8k"}],"languages":[{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["GSM8K","GSM8k (5-shot)","GSM8k TR v0.2","GSM8k TR","gsm8k (5-shots)"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/TozluLider6393/deneme5","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/gsm8k","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/juletxara/mgsm","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/d0rj/gsm8k-ru","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/malhajar/gsm8k-tr","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/openai/gsm8k","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/gretelai/gsm8k-synthetic-diverse-8b","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/jtlucas/test7","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/bezir/gsm8k-tr","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/neurotechnology/lt_gsm8k","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/gretelai/gsm8k-synthetic-diverse-405b","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/gretelai/synthetic-gsm8k-reflection-405b","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/gretelai/synthetic-gsm8k-evolutionary-405b","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/10nates/synthetic-gsm8k-reflection-405b-autotrain-ready","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/1-800-SHARED-TASKS/openai-gsm8k","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Hindi-Gemma/openai-gsm8k","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/gretelai/gretel-math-gsm8k-v0","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/oieieio/gsm8k","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Satyam-Singh/logical","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/revonodes/gsm8k","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/yarinkos/first-dataset","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/gsm8k","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":1881,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/arithmetic-reasoning-on-gsm8k","task":"Arithmetic Reasoning","dataset_variant":"GSM8K","rows":164,"metrics":["Accuracy","Parameters (Billion)"],"first_row_in_archive_order":{"model":"Claude 3.5 Sonnet (HPT)","paper":"/paper/hierarchical-prompting-taxonomy-a-universal","metrics":{"Accuracy":"97.72"},"code_links":[{"title":"devichand579/HPT","url":"https://github.com/devichand579/HPT"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/gsm8k-on-gsm8k","task":"GSM8K","dataset_variant":"GSM8K","rows":3,"metrics":["Accuracy","0-shot MRR"],"first_row_in_archive_order":{"model":"Xolver","paper":"/paper/xolver-multi-agent-reasoning-with-holistic","metrics":{"Accuracy":"98.1"},"code_links":[{"title":"kagnlp/Xolver","url":"https://github.com/kagnlp/Xolver"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/gsm8k-on-gsm8k-5-shots","task":"GSM8K","dataset_variant":"gsm8k (5-shots)","rows":0,"metrics":["exact_match_strict"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/mathematical-reasoning-on-gsm8k","task":"Mathematical Reasoning","dataset_variant":"GSM8K","rows":0,"metrics":["Accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-gsm8k-5-shot","task":"Text Generation","dataset_variant":"GSM8k (5-shot)","rows":0,"metrics":["accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-gsm8k-tr","task":"Text Generation","dataset_variant":"GSM8k TR","rows":0,"metrics":["accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-gsm8k-tr-v0-2","task":"Text Generation","dataset_variant":"GSM8k TR v0.2","rows":0,"metrics":["accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/xolver-multi-agent-reasoning-with-holistic","title":"Xolver: Multi-Agent Reasoning with Holistic Experience Learning Just Like an Olympiad Team","date":"2025-06-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/capo-cost-aware-prompt-optimization","title":"CAPO: Cost-Aware Prompt Optimization","date":"2025-04-22","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/mygo-multiplex-cot-a-method-for-self","title":"MyGO Multiplex CoT: A Method for Self-Reflection in Large Language Models via Double Chain of Thought Thinking","date":"2025-01-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/teaching-inspired-integrated-prompting","title":"Teaching-Inspired Integrated Prompting Framework: A Novel Approach for Enhancing Reasoning in Large Language Models","date":"2024-10-10","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":8,"samples_unverified":0,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/openmathinstruct-2-accelerating-ai-for-math","title":"OpenMathInstruct-2: Accelerating AI for Math with Massive Open-Source Instruction Data","date":"2024-10-02","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":0,"samples_unverified":15,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/qwen2-technical-report","title":"Qwen2 Technical Report","date":"2024-07-15","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/step-dpo-step-wise-preference-optimization","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","date":"2024-06-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":7,"samples_unverified":5,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hierarchical-prompting-taxonomy-a-universal","title":"Hierarchical Prompting Taxonomy: A Universal Evaluation Framework for Large Language Models Aligned with Human Cognitive Principles","date":"2024-06-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dart-math-difficulty-aware-rejection-tuning-1","title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","date":"2024-06-18","rows_on_this_dataset":8,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":10,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/breaking-the-ceiling-of-the-llm-community-by","title":"Breaking the Ceiling of the LLM Community by Treating Token Generation as a Classification for Ensembling","date":"2024-06-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/achieving-97-on-gsm8k-deeply-understanding","title":"Achieving >97% on GSM8K: Deeply Understanding the Problems Makes LLMs Better Solvers for Math Word Problems","date":"2024-04-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/toward-self-improvement-of-llms-via","title":"Toward Self-Improvement of LLMs via Imagination, Searching, and Criticizing","date":"2024-04-18","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":13,"samples_unverified":5,"pointer_only_for_licence":18,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/branch-train-mix-mixing-expert-llms-into-a","title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","date":"2024-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/the-claude-3-model-family-opus-sonnet-haiku","title":"The Claude 3 Model Family: Opus, Sonnet, Haiku","date":"2024-03-04","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-data-ability-boundary","title":"An Empirical Study of Data Ability Boundary in LLMs' Math Reasoning","date":"2024-02-23","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":9,"samples_unverified":2,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/orca-math-unlocking-the-potential-of-slms-in","title":"Orca-Math: Unlocking the potential of SLMs in Grade School Math","date":"2024-02-16","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/openmathinstruct-1-a-1-8-million-math","title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","date":"2024-02-15","rows_on_this_dataset":12,"code_links":1,"syntology":null},{"paper":"/paper/the-unreasonable-effectiveness-of-eccentric","title":"The Unreasonable Effectiveness of Eccentric Automatic Prompts","date":"2024-02-09","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/deepseekmath-pushing-the-limits-of","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","date":"2024-02-05","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":8,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/parameter-efficient-sparsity-crafting-from","title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","date":"2024-01-05","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/gemini-a-family-of-highly-capable-multimodal-1","title":"Gemini: A Family of Highly Capable Multimodal Models","date":"2023-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tinygsm-achieving-80-on-gsm8k-with-small","title":"TinyGSM: achieving >80% on GSM8k with small language models","date":"2023-12-14","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/math-shepherd-a-label-free-step-by-step","title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","date":"2023-12-14","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/boosting-llm-reasoning-push-the-limits-of-few","title":"Fewer is More: Boosting LLM Reasoning with Reinforced Context Pruning","date":"2023-12-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/frugal-lms-trained-to-invoke-symbolic-solvers","title":"Frugal LMs Trained to Invoke Symbolic Solvers Achieve Parameter-Efficient Arithmetic Reasoning","date":"2023-12-09","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/orca-2-teaching-small-language-models-how-to","title":"Orca 2: Teaching Small Language Models How to Reason","date":"2023-11-18","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/outcome-supervised-verifiers-for-planning-in","title":"OVM, Outcome-supervised Value Models for Planning in Mathematical Reasoning","date":"2023-11-16","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/the-art-of-llm-refinement-ask-refine-and","title":"The ART of LLM Refinement: Ask, Refine, and Trust","date":"2023-11-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/llemma-an-open-language-model-for-mathematics","title":"Llemma: An Open Language Model For Mathematics","date":"2023-10-16","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/kwaiyiimath-technical-report","title":"KwaiYiiMath: Technical Report","date":"2023-10-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/mistral-7b","title":"Mistral 7B","date":"2023-10-10","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":9,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/query-and-response-augmentation-cannot-help","title":"MuggleMath: Assessing the Impact of Query and Response Augmentation on Math Reasoning","date":"2023-10-09","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/mathcoder-seamless-code-integration-in-llms","title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","date":"2023-10-05","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tora-a-tool-integrated-reasoning-agent-for","title":"ToRA: A Tool-Integrated Reasoning Agent for Mathematical Problem Solving","date":"2023-09-29","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":5,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/metamath-bootstrap-your-own-mathematical","title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","date":"2023-09-21","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":22,"samples_ran":15,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/openchat-advancing-open-source-language","title":"OpenChat: Advancing Open-source Language Models with Mixed-Quality Data","date":"2023-09-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/wizardmath-empowering-mathematical-reasoning","title":"WizardMath: Empowering Mathematical Reasoning for Large Language Models via Reinforced Evol-Instruct","date":"2023-08-18","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":10,"samples_unverified":6,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scaling-relationship-on-learning-mathematical","title":"Scaling Relationship on Learning Mathematical Reasoning with Large Language Models","date":"2023-08-03","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llama-2-open-foundation-and-fine-tuned-chat","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","date":"2023-07-18","rows_on_this_dataset":1,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":52,"samples_ran":31,"samples_unverified":21,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/model-card-and-evaluations-for-claude-models","title":"Model Card and Evaluations for Claude Models","date":"2023-07-11","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/palm-2-technical-report-1","title":"PaLM 2 Technical Report","date":"2023-05-17","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/codet5-open-code-large-language-models-for","title":"CodeT5+: Open Code Large Language Models for Code Understanding and Generation","date":"2023-05-13","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sparks-of-artificial-general-intelligence","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","date":"2023-03-22","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","rows_on_this_dataset":8,"code_links":57,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":58,"samples_ran":26,"samples_unverified":32,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lever-learning-to-verify-language-to-code","title":"LEVER: Learning to Verify Language-to-Code Generation with Execution","date":"2023-02-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":22,"samples_ran":6,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/solving-math-word-problems-with-process-and","title":"Solving math word problems with process- and outcome-based feedback","date":"2022-11-25","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/solving-math-word-problem-via-cooperative","title":"Solving Math Word Problems via Cooperative Reasoning induced Language Models","date":"2022-10-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/transcending-scaling-laws-with-0-1-extra","title":"Transcending Scaling Laws with 0.1% Extra Compute","date":"2022-10-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/large-language-models-can-self-improve","title":"Large Language Models Can Self-Improve","date":"2022-10-20","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/composing-ensembles-of-pre-trained-models-via","title":"Composing Ensembles of Pre-trained Models via Iterative Consensus","date":"2022-10-20","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/solving-quantitative-reasoning-problems-with","title":"Solving Quantitative Reasoning Problems with Language Models","date":"2022-06-29","rows_on_this_dataset":10,"code_links":1,"syntology":null},{"paper":"/paper/on-the-advance-of-making-language-models","title":"Making Large Language Models Better Reasoners with Step-Aware Verifier","date":"2022-06-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-from-self-sampled-correct-and","title":"Learning Math Reasoning from Self-Sampled Correct and Partially-Correct Solutions","date":"2022-05-28","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":0,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/large-language-models-are-zero-shot-reasoners","title":"Large Language Models are Zero-Shot Reasoners","date":"2022-05-24","rows_on_this_dataset":7,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/least-to-most-prompting-enables-complex","title":"Least-to-Most Prompting Enables Complex Reasoning in Large Language Models","date":"2022-05-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unifying-language-learning-paradigms","title":"UL2: Unifying Language Learning Paradigms","date":"2022-05-10","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":0,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-consistency-improves-chain-of-thought","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","date":"2022-03-21","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":28,"samples_harvested":370,"samples_ran":183,"samples_unverified":187,"pointer_only_for_licence":96,"papers_with_no_sample_that_ran":6,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}