{"url":"/sota/arithmetic-reasoning-on-gsm8k","task":{"name":"Arithmetic Reasoning","url":"/task/arithmetic-reasoning","note":null},"dataset":{"name":"GSM8K","url":"/dataset/gsm8k"},"category":"Reasoning","categories":["Reasoning"],"category_note":null,"description":null,"description_from":null,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["Accuracy","Parameters (Billion)"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"Accuracy":"higher","Parameters (Billion)":null}},"counts":{"rows":164,"rows_with_code":118,"rows_with_paper_page":150,"rows_dated":150,"rows_using_additional_data":85},"rows":[{"rank_in_archive_order":1,"model":"Claude 3.5 Sonnet (HPT)","metrics":{"Accuracy":"97.72"},"uses_additional_data":false,"paper_date":"2024-06-18","paper":"/paper/hierarchical-prompting-taxonomy-a-universal","paper_url":"https://arxiv.org/abs/2406.12644v4","paper_title":"Hierarchical Prompting Taxonomy: A Universal Evaluation Framework for Large Language Models Aligned with Human Cognitive Principles","code":"https://github.com/devichand579/HPT","n_code_links":1,"syntology":null},{"rank_in_archive_order":2,"model":"DUP prompt upon GPT-4","metrics":{"Accuracy":"97.1"},"uses_additional_data":false,"paper_date":"2024-04-23","paper":"/paper/achieving-97-on-gsm8k-deeply-understanding","paper_url":"https://arxiv.org/abs/2404.14963v5","paper_title":"Achieving >97% on GSM8K: Deeply Understanding the Problems Makes LLMs Better Solvers for Math Word Problems","code":"https://github.com/whu-zqh/dup","n_code_links":1,"syntology":null},{"rank_in_archive_order":3,"model":"Qwen2-Math-72B-Instruct\n(greedy)","metrics":{"Accuracy":"96.7","Parameters (Billion)":"72"},"uses_additional_data":true,"paper_date":"2024-07-15","paper":"/paper/qwen2-technical-report","paper_url":"https://arxiv.org/abs/2407.10671v4","paper_title":"Qwen2 Technical Report","code":"https://github.com/qwenlm/qwen1.5","n_code_links":6,"syntology":null},{"rank_in_archive_order":4,"model":"SFT-Mistral-7B (Metamath, OVM, Smart Ensemble)","metrics":{"Accuracy":"96.4","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":5,"model":"OpenMath2-Llama3.1-70B (majority@256)","metrics":{"Accuracy":"96.0"},"uses_additional_data":true,"paper_date":"2024-10-02","paper":"/paper/openmathinstruct-2-accelerating-ai-for-math","paper_url":"https://arxiv.org/abs/2410.01560v2","paper_title":"OpenMathInstruct-2: Accelerating AI for Math with Massive Open-Source Instruction Data","code":"https://github.com/NVIDIA/NeMo-Skills","n_code_links":1,"syntology":{"n_ran":12,"n_unverified":3,"n_samples":15,"n_pointer_only_licence":0}},{"rank_in_archive_order":6,"model":"Jiutian-大模型","metrics":{"Accuracy":"95.2","Parameters (Billion)":"75"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":7,"model":"DAMOMath-7B(MetaMath, OVM, BS, Ensemble)","metrics":{"Accuracy":"95.1","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":8,"model":"Claude 3 Opus (0-shot chain-of-thought)","metrics":{"Accuracy":"95"},"uses_additional_data":false,"paper_date":"2024-03-04","paper":"/paper/the-claude-3-model-family-opus-sonnet-haiku","paper_url":"https://www.anthropic.com/news/claude-3-family","paper_title":"The Claude 3 Model Family: Opus, Sonnet, Haiku","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":9,"model":"OpenMath2-Llama3.1-70B","metrics":{"Accuracy":"94.9"},"uses_additional_data":true,"paper_date":"2024-10-02","paper":"/paper/openmathinstruct-2-accelerating-ai-for-math","paper_url":"https://arxiv.org/abs/2410.01560v2","paper_title":"OpenMathInstruct-2: Accelerating AI for Math with Massive Open-Source Instruction Data","code":"https://github.com/NVIDIA/NeMo-Skills","n_code_links":1,"syntology":{"n_ran":12,"n_unverified":3,"n_samples":15,"n_pointer_only_licence":0}},{"rank_in_archive_order":10,"model":"GPT-4 (Teaching-Inspired)","metrics":{"Accuracy":"94.8"},"uses_additional_data":false,"paper_date":"2024-10-10","paper":"/paper/teaching-inspired-integrated-prompting","paper_url":"https://arxiv.org/abs/2410.08068v1","paper_title":"Teaching-Inspired Integrated Prompting Framework: A Novel Approach for Enhancing Reasoning in Large Language Models","code":"https://github.com/sallytan13/teaching-inspired-prompting","n_code_links":1,"syntology":{"n_ran":8,"n_unverified":0,"n_samples":8,"n_pointer_only_licence":8}},{"rank_in_archive_order":11,"model":"SFT-Mistral-7B (Metamath + ovm +ensemble)","metrics":{"Accuracy":"94.13","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":12,"model":"OpenMath2-Llama3.1-8B (majority@256)","metrics":{"Accuracy":"94.1"},"uses_additional_data":true,"paper_date":"2024-10-02","paper":"/paper/openmathinstruct-2-accelerating-ai-for-math","paper_url":"https://arxiv.org/abs/2410.01560v2","paper_title":"OpenMathInstruct-2: Accelerating AI for Math with Massive Open-Source Instruction Data","code":"https://github.com/NVIDIA/NeMo-Skills","n_code_links":1,"syntology":{"n_ran":12,"n_unverified":3,"n_samples":15,"n_pointer_only_licence":0}},{"rank_in_archive_order":13,"model":"Qwen2-72B-Instruct-Step-DPO (0-shot CoT)","metrics":{"Accuracy":"94.0"},"uses_additional_data":true,"paper_date":"2024-06-26","paper":"/paper/step-dpo-step-wise-preference-optimization","paper_url":"https://arxiv.org/abs/2406.18629v1","paper_title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","code":"https://github.com/dvlab-research/step-dpo","n_code_links":1,"syntology":{"n_ran":7,"n_unverified":5,"n_samples":12,"n_pointer_only_licence":12}},{"rank_in_archive_order":14,"model":"DAMOMath-7B(MetaMath, OVM, Ensemble)","metrics":{"Accuracy":"93.2","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":15,"model":"Claude 3 Sonnet (0-shot chain-of-thought)","metrics":{"Accuracy":"92.3"},"uses_additional_data":false,"paper_date":"2024-03-04","paper":"/paper/the-claude-3-model-family-opus-sonnet-haiku","paper_url":"https://www.anthropic.com/news/claude-3-family","paper_title":"The Claude 3 Model Family: Opus, Sonnet, Haiku","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":16,"model":"AlphaLLM (with MCTS)","metrics":{"Accuracy":"92","Parameters (Billion)":"70"},"uses_additional_data":false,"paper_date":"2024-04-18","paper":"/paper/toward-self-improvement-of-llms-via","paper_url":"https://arxiv.org/abs/2404.12253v2","paper_title":"Toward Self-Improvement of LLMs via Imagination, Searching, and Criticizing","code":"https://github.com/yetianjhu/alphallm","n_code_links":1,"syntology":{"n_ran":13,"n_unverified":5,"n_samples":18,"n_pointer_only_licence":18}},{"rank_in_archive_order":17,"model":"OpenMath2-Llama3.1-8B","metrics":{"Accuracy":"91.7"},"uses_additional_data":true,"paper_date":"2024-10-02","paper":"/paper/openmathinstruct-2-accelerating-ai-for-math","paper_url":"https://arxiv.org/abs/2410.01560v2","paper_title":"OpenMathInstruct-2: Accelerating AI for Math with Massive Open-Source Instruction Data","code":"https://github.com/NVIDIA/NeMo-Skills","n_code_links":1,"syntology":{"n_ran":12,"n_unverified":3,"n_samples":15,"n_pointer_only_licence":0}},{"rank_in_archive_order":18,"model":"PaLM 2 (few-shot, k=8, SC)","metrics":{"Accuracy":"91.0"},"uses_additional_data":false,"paper_date":"2023-05-17","paper":"/paper/palm-2-technical-report-1","paper_url":"https://arxiv.org/abs/2305.10403v3","paper_title":"PaLM 2 Technical Report","code":"https://github.com/eternityyw/tram-benchmark","n_code_links":1,"syntology":null},{"rank_in_archive_order":19,"model":"GaC(Qwen2-72B-Instruct + Llama-3-70B-Instruct)","metrics":{"Accuracy":"90.91"},"uses_additional_data":false,"paper_date":"2024-06-18","paper":"/paper/breaking-the-ceiling-of-the-llm-community-by","paper_url":"https://arxiv.org/abs/2406.12585v2","paper_title":"Breaking the Ceiling of the LLM Community by Treating Token Generation as a Classification for Ensembling","code":"https://github.com/yaoching0/gac","n_code_links":1,"syntology":{"n_ran":4,"n_unverified":3,"n_samples":7,"n_pointer_only_licence":0}},{"rank_in_archive_order":20,"model":"OpenMath-CodeLlama-70B (w/ code, SC, k=50)","metrics":{"Accuracy":"90.8","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2024-02-15","paper":"/paper/openmathinstruct-1-a-1-8-million-math","paper_url":"https://arxiv.org/abs/2402.10176v2","paper_title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","code":"https://github.com/kipok/nemo-skills","n_code_links":1,"syntology":null},{"rank_in_archive_order":21,"model":"DART-Math-Llama3-70B-Uniform (0-shot CoT, w/o code)","metrics":{"Accuracy":"90.4","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2024-06-18","paper":"/paper/dart-math-difficulty-aware-rejection-tuning-1","paper_url":"https://arxiv.org/abs/2407.13690v2","paper_title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","code":"https://github.com/hkust-nlp/dart-math","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":2,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":22,"model":"OpenMath-Llama2-70B (w/ code, SC, k=50)","metrics":{"Accuracy":"90.1","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2024-02-15","paper":"/paper/openmathinstruct-1-a-1-8-million-math","paper_url":"https://arxiv.org/abs/2402.10176v2","paper_title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","code":"https://github.com/kipok/nemo-skills","n_code_links":1,"syntology":null},{"rank_in_archive_order":23,"model":"DART-Math-Llama3-70B-Prop2Diff (0-shot CoT, w/o code)","metrics":{"Accuracy":"89.6","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2024-06-18","paper":"/paper/dart-math-difficulty-aware-rejection-tuning-1","paper_url":"https://arxiv.org/abs/2407.13690v2","paper_title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","code":"https://github.com/hkust-nlp/dart-math","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":2,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":24,"model":"Shepherd+Mistral-7B (SFT on MetaMATH + PRM RL+ PRM rerank, k=256)","metrics":{"Accuracy":"89.1","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2023-12-14","paper":"/paper/math-shepherd-a-label-free-step-by-step","paper_url":"https://arxiv.org/abs/2312.08935v3","paper_title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","code":"https://github.com/hkust-nlp/b-star","n_code_links":3,"syntology":null},{"rank_in_archive_order":25,"model":"Llama SFT (Metamath ToRA Ensemble)","metrics":{"Accuracy":"89.0","Parameters (Billion)":"13"},"uses_additional_data":true,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":26,"model":"Minerva 62B (maj5@100)","metrics":{"Accuracy":"89","Parameters (Billion)":"62"},"uses_additional_data":false,"paper_date":"2022-06-29","paper":"/paper/solving-quantitative-reasoning-problems-with","paper_url":"https://arxiv.org/abs/2206.14858v2","paper_title":"Solving Quantitative Reasoning Problems with Language Models","code":"https://github.com/gair-nlp/abel","n_code_links":1,"syntology":null},{"rank_in_archive_order":27,"model":"Claude 3 Haiku (0-shot chain-of-thought)","metrics":{"Accuracy":"88.9"},"uses_additional_data":false,"paper_date":"2024-03-04","paper":"/paper/the-claude-3-model-family-opus-sonnet-haiku","paper_url":"https://www.anthropic.com/news/claude-3-family","paper_title":"The Claude 3 Model Family: Opus, Sonnet, Haiku","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":28,"model":"ToRA-70B (SC, k=50)","metrics":{"Accuracy":"88.3","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2023-09-29","paper":"/paper/tora-a-tool-integrated-reasoning-agent-for","paper_url":"https://arxiv.org/abs/2309.17452v4","paper_title":"ToRA: A Tool-Integrated Reasoning Agent for Mathematical Problem Solving","code":"https://github.com/microsoft/tora","n_code_links":1,"syntology":{"n_ran":6,"n_unverified":6,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":29,"model":"DeepSeekMATH-RL-7B","metrics":{"Accuracy":"88.2","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2024-02-05","paper":"/paper/deepseekmath-pushing-the-limits-of","paper_url":"https://arxiv.org/abs/2402.03300v3","paper_title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","code":"https://github.com/shibing624/medicalgpt","n_code_links":5,"syntology":{"n_ran":10,"n_unverified":14,"n_samples":24,"n_pointer_only_licence":0}},{"rank_in_archive_order":30,"model":"DART-Math-DSMath-7B-Uniform (0-shot CoT, w/o code)","metrics":{"Accuracy":"88.2","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2024-06-18","paper":"/paper/dart-math-difficulty-aware-rejection-tuning-1","paper_url":"https://arxiv.org/abs/2407.13690v2","paper_title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","code":"https://github.com/hkust-nlp/dart-math","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":2,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":31,"model":"OpenMath-CodeLlama-34B (w/ code, SC, k=50)","metrics":{"Accuracy":"88.0","Parameters (Billion)":"34"},"uses_additional_data":true,"paper_date":"2024-02-15","paper":"/paper/openmathinstruct-1-a-1-8-million-math","paper_url":"https://arxiv.org/abs/2402.10176v2","paper_title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","code":"https://github.com/kipok/nemo-skills","n_code_links":1,"syntology":null},{"rank_in_archive_order":32,"model":"Claude 2 (0-shot chain-of-thought)","metrics":{"Accuracy":"88"},"uses_additional_data":false,"paper_date":"2023-07-11","paper":"/paper/model-card-and-evaluations-for-claude-models","paper_url":"https://www-files.anthropic.com/production/images/Model-Card-Claude-2.pdf","paper_title":"Model Card and Evaluations for Claude Models","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":33,"model":"Shivaay-4B (8-shot chain-of-thought)","metrics":{"Accuracy":"87.41","Parameters (Billion)":"4"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":34,"model":"DeepMind 70B Model (SFT+ORM-RL, ORM reranking)","metrics":{"Accuracy":"87.3","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2022-11-25","paper":"/paper/solving-math-word-problems-with-process-and","paper_url":"https://arxiv.org/abs/2211.14275v1","paper_title":"Solving math word problems with process- and outcome-based feedback","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":35,"model":"MMOS-DeepSeekMath-7B(0-shot,k=50)","metrics":{"Accuracy":"87.2","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2024-02-23","paper":"/paper/an-empirical-study-of-data-ability-boundary","paper_url":"https://arxiv.org/abs/2403.00799v1","paper_title":"An Empirical Study of Data Ability Boundary in LLMs' Math Reasoning","code":"https://github.com/cyzhh/MMOS","n_code_links":1,"syntology":{"n_ran":9,"n_unverified":2,"n_samples":11,"n_pointer_only_licence":11}},{"rank_in_archive_order":36,"model":"DeepMind 70B Model (SFT+PRM-RL, PRM reranking)","metrics":{"Accuracy":"87.1","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2022-11-25","paper":"/paper/solving-math-word-problems-with-process-and","paper_url":"https://arxiv.org/abs/2211.14275v1","paper_title":"Solving math word problems with process- and outcome-based feedback","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":37,"model":"GPT-4","metrics":{"Accuracy":"87.1"},"uses_additional_data":false,"paper_date":"2023-03-22","paper":"/paper/sparks-of-artificial-general-intelligence","paper_url":"https://arxiv.org/abs/2303.12712v5","paper_title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","code":"https://github.com/microsoft/guidance","n_code_links":3,"syntology":{"n_ran":9,"n_unverified":1,"n_samples":10,"n_pointer_only_licence":0}},{"rank_in_archive_order":38,"model":"OpenMath-Mistral-7B (w/ code, SC, k=50)","metrics":{"Accuracy":"86.9","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2024-02-15","paper":"/paper/openmathinstruct-1-a-1-8-million-math","paper_url":"https://arxiv.org/abs/2402.10176v2","paper_title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","code":"https://github.com/kipok/nemo-skills","n_code_links":1,"syntology":null},{"rank_in_archive_order":39,"model":"Orca-Math 7B (fine-tuned)","metrics":{"Accuracy":"86.8","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2024-02-16","paper":"/paper/orca-math-unlocking-the-potential-of-slms-in","paper_url":"https://arxiv.org/abs/2402.14830v1","paper_title":"Orca-Math: Unlocking the potential of SLMs in Grade School Math","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":40,"model":"DART-Math-DSMath-7B-Prop2Diff (0-shot CoT, w/o code)","metrics":{"Accuracy":"86.8","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2024-06-18","paper":"/paper/dart-math-difficulty-aware-rejection-tuning-1","paper_url":"https://arxiv.org/abs/2407.13690v2","paper_title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","code":"https://github.com/hkust-nlp/dart-math","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":2,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":41,"model":"OpenMath-CodeLlama-13B (w/ code, SC, k=50)","metrics":{"Accuracy":"86.8","Parameters (Billion)":"13"},"uses_additional_data":true,"paper_date":"2024-02-15","paper":"/paper/openmathinstruct-1-a-1-8-million-math","paper_url":"https://arxiv.org/abs/2402.10176v2","paper_title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","code":"https://github.com/kipok/nemo-skills","n_code_links":1,"syntology":null},{"rank_in_archive_order":42,"model":"Gemini Pro (maj1@32)","metrics":{"Accuracy":"86.5"},"uses_additional_data":false,"paper_date":"2023-12-19","paper":"/paper/gemini-a-family-of-highly-capable-multimodal-1","paper_url":"https://arxiv.org/abs/2312.11805v5","paper_title":"Gemini: A Family of Highly Capable Multimodal Models","code":"https://github.com/valdecy/pybibx","n_code_links":1,"syntology":null},{"rank_in_archive_order":43,"model":"Codex (Self-Evaluation Guided Decoding, PAL, multiple reasoning chains, 9-shot gen, 5-shot eval)","metrics":{"Accuracy":"85.5"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":44,"model":"Claude 1.3 (0-shot chain-of-thought)","metrics":{"Accuracy":"85.2"},"uses_additional_data":false,"paper_date":"2023-07-11","paper":"/paper/model-card-and-evaluations-for-claude-models","paper_url":"https://www-files.anthropic.com/production/images/Model-Card-Claude-2.pdf","paper_title":"Model Card and Evaluations for Claude Models","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":45,"model":"ToRA-Code-34B (SC, k=50)","metrics":{"Accuracy":"85.1","Parameters (Billion)":"34"},"uses_additional_data":true,"paper_date":"2023-09-29","paper":"/paper/tora-a-tool-integrated-reasoning-agent-for","paper_url":"https://arxiv.org/abs/2309.17452v4","paper_title":"ToRA: A Tool-Integrated Reasoning Agent for Mathematical Problem Solving","code":"https://github.com/microsoft/tora","n_code_links":1,"syntology":{"n_ran":6,"n_unverified":6,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":46,"model":"OpenMath-CodeLlama-7B (w/ code, SC, k=50)","metrics":{"Accuracy":"84.8","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2024-02-15","paper":"/paper/openmathinstruct-1-a-1-8-million-math","paper_url":"https://arxiv.org/abs/2402.10176v2","paper_title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","code":"https://github.com/kipok/nemo-skills","n_code_links":1,"syntology":null},{"rank_in_archive_order":47,"model":"OVM-Mistral-7B (verify100@1)","metrics":{"Accuracy":"84.7","Parameters (Billion)":"7"},"uses_additional_data":false,"paper_date":"2023-11-16","paper":"/paper/outcome-supervised-verifiers-for-planning-in","paper_url":"https://arxiv.org/abs/2311.09724v2","paper_title":"OVM, Outcome-supervised Value Models for Planning in Mathematical Reasoning","code":"https://github.com/freedomintelligence/ovm","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":0,"n_samples":3,"n_pointer_only_licence":3}},{"rank_in_archive_order":48,"model":"OpenMath-Llama2-70B (w/ code)","metrics":{"Accuracy":"84.7","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2024-02-15","paper":"/paper/openmathinstruct-1-a-1-8-million-math","paper_url":"https://arxiv.org/abs/2402.10176v2","paper_title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","code":"https://github.com/kipok/nemo-skills","n_code_links":1,"syntology":null},{"rank_in_archive_order":49,"model":"OpenMath-CodeLlama-70B (w/ code)","metrics":{"Accuracy":"84.6","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2024-02-15","paper":"/paper/openmathinstruct-1-a-1-8-million-math","paper_url":"https://arxiv.org/abs/2402.10176v2","paper_title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","code":"https://github.com/kipok/nemo-skills","n_code_links":1,"syntology":null},{"rank_in_archive_order":50,"model":"code-davinci-002 175B (LEVER, 8-shot)","metrics":{"Accuracy":"84.5","Parameters (Billion)":"175"},"uses_additional_data":false,"paper_date":"2023-02-16","paper":"/paper/lever-learning-to-verify-language-to-code","paper_url":"https://arxiv.org/abs/2302.08468v3","paper_title":"LEVER: Learning to Verify Language-to-Code Generation with Execution","code":"https://github.com/niansong1996/lever","n_code_links":1,"syntology":{"n_ran":18,"n_unverified":4,"n_samples":22,"n_pointer_only_licence":0}},{"rank_in_archive_order":51,"model":"ToRA 70B","metrics":{"Accuracy":"84.3","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2023-09-29","paper":"/paper/tora-a-tool-integrated-reasoning-agent-for","paper_url":"https://arxiv.org/abs/2309.17452v4","paper_title":"ToRA: A Tool-Integrated Reasoning Agent for Mathematical Problem Solving","code":"https://github.com/microsoft/tora","n_code_links":1,"syntology":{"n_ran":6,"n_unverified":6,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":52,"model":"Shepherd + Mistral-7B (SFT on MetaMATH + PRM RL)","metrics":{"Accuracy":"84.1","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2023-12-14","paper":"/paper/math-shepherd-a-label-free-step-by-step","paper_url":"https://arxiv.org/abs/2312.08935v3","paper_title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","code":"https://github.com/hkust-nlp/b-star","n_code_links":3,"syntology":null},{"rank_in_archive_order":53,"model":"MathCoder-L-70B","metrics":{"Accuracy":"83.9","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2023-10-05","paper":"/paper/mathcoder-seamless-code-integration-in-llms","paper_url":"https://arxiv.org/abs/2310.03731v1","paper_title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","code":"https://github.com/mathllm/mathcoder","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":0}},{"rank_in_archive_order":54,"model":"WizardMath-7B-V1.1","metrics":{"Accuracy":"83.2","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2023-08-18","paper":"/paper/wizardmath-empowering-mathematical-reasoning","paper_url":"https://arxiv.org/abs/2308.09583v2","paper_title":"WizardMath: Empowering Mathematical Reasoning for Large Language Models via Reinforced Evol-Instruct","code":"https://github.com/nlpxucan/wizardlm","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":6,"n_samples":16,"n_pointer_only_licence":16}},{"rank_in_archive_order":55,"model":"DIVERSE 175B (8-shot)","metrics":{"Accuracy":"83.2","Parameters (Billion)":"175"},"uses_additional_data":false,"paper_date":"2022-06-06","paper":"/paper/on-the-advance-of-making-language-models","paper_url":"https://arxiv.org/abs/2206.02336v3","paper_title":"Making Large Language Models Better Reasoners with Step-Aware Verifier","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":56,"model":"OVM-Mistral-7B (verify20@1)","metrics":{"Accuracy":"82.6","Parameters (Billion)":"7"},"uses_additional_data":false,"paper_date":"2023-11-16","paper":"/paper/outcome-supervised-verifiers-for-planning-in","paper_url":"https://arxiv.org/abs/2311.09724v2","paper_title":"OVM, Outcome-supervised Value Models for Planning in Mathematical Reasoning","code":"https://github.com/freedomintelligence/ovm","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":0,"n_samples":3,"n_pointer_only_licence":3}},{"rank_in_archive_order":57,"model":"DART-Math-Mistral-7B-Uniform (0-shot CoT, w/o code)","metrics":{"Accuracy":"82.6","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2024-06-18","paper":"/paper/dart-math-difficulty-aware-rejection-tuning-1","paper_url":"https://arxiv.org/abs/2407.13690v2","paper_title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","code":"https://github.com/hkust-nlp/dart-math","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":2,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":58,"model":"ChatGPT (Ask, Refine, Trust)","metrics":{"Accuracy":"82.6"},"uses_additional_data":false,"paper_date":"2023-11-14","paper":"/paper/the-art-of-llm-refinement-ask-refine-and","paper_url":"https://arxiv.org/abs/2311.07961v1","paper_title":"The ART of LLM Refinement: Ask, Refine, and Trust","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":59,"model":"DART-Math-Llama3-8B-Uniform (0-shot CoT, w/o code)","metrics":{"Accuracy":"82.5","Parameters (Billion)":"8"},"uses_additional_data":true,"paper_date":"2024-06-18","paper":"/paper/dart-math-difficulty-aware-rejection-tuning-1","paper_url":"https://arxiv.org/abs/2407.13690v2","paper_title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","code":"https://github.com/hkust-nlp/dart-math","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":2,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":60,"model":"MetaMath 70B","metrics":{"Accuracy":"82.3","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2023-09-21","paper":"/paper/metamath-bootstrap-your-own-mathematical","paper_url":"https://arxiv.org/abs/2309.12284v4","paper_title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","code":"https://github.com/meta-math/MetaMath","n_code_links":1,"syntology":{"n_ran":15,"n_unverified":7,"n_samples":22,"n_pointer_only_licence":0}},{"rank_in_archive_order":61,"model":"MuggleMATH 70B","metrics":{"Accuracy":"82.3","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2023-10-09","paper":"/paper/query-and-response-augmentation-cannot-help","paper_url":"https://arxiv.org/abs/2310.05506v3","paper_title":"MuggleMath: Assessing the Impact of Query and Response Augmentation on Math Reasoning","code":"https://github.com/ofa-sys/gsm8k-screl","n_code_links":1,"syntology":null},{"rank_in_archive_order":62,"model":"PaLM 540B (Self Improvement, Self Consistency)","metrics":{"Accuracy":"82.1","Parameters (Billion)":"540"},"uses_additional_data":false,"paper_date":"2022-10-20","paper":"/paper/large-language-models-can-self-improve","paper_url":"https://arxiv.org/abs/2210.11610v2","paper_title":"Large Language Models Can Self-Improve","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":63,"model":"MathCoder-CL-34B","metrics":{"Accuracy":"81.7","Parameters (Billion)":"34"},"uses_additional_data":true,"paper_date":"2023-10-05","paper":"/paper/mathcoder-seamless-code-integration-in-llms","paper_url":"https://arxiv.org/abs/2310.03731v1","paper_title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","code":"https://github.com/mathllm/mathcoder","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":0}},{"rank_in_archive_order":64,"model":"WizardMath-70B-V1.0","metrics":{"Accuracy":"81.6","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2023-08-18","paper":"/paper/wizardmath-empowering-mathematical-reasoning","paper_url":"https://arxiv.org/abs/2308.09583v2","paper_title":"WizardMath: Empowering Mathematical Reasoning for Large Language Models via Reinforced Evol-Instruct","code":"https://github.com/nlpxucan/wizardlm","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":6,"n_samples":16,"n_pointer_only_licence":16}},{"rank_in_archive_order":65,"model":"Phi-GSM+V 1.3B+1.3B (verify48@1)","metrics":{"Accuracy":"81.5","Parameters (Billion)":"2.6"},"uses_additional_data":false,"paper_date":"2023-12-14","paper":"/paper/tinygsm-achieving-80-on-gsm8k-with-small","paper_url":"https://arxiv.org/abs/2312.09241v1","paper_title":"TinyGSM: achieving >80% on GSM8k with small language models","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":66,"model":"DART-Math-Mistral-7B-Prop2Diff (0-shot CoT, w/o code)","metrics":{"Accuracy":"81.1","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2024-06-18","paper":"/paper/dart-math-difficulty-aware-rejection-tuning-1","paper_url":"https://arxiv.org/abs/2407.13690v2","paper_title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","code":"https://github.com/hkust-nlp/dart-math","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":2,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":67,"model":"DART-Math-Llama3-8B-Prop2Diff (0-shot CoT, w/o code)","metrics":{"Accuracy":"81.1","Parameters (Billion)":"8"},"uses_additional_data":true,"paper_date":"2024-06-18","paper":"/paper/dart-math-difficulty-aware-rejection-tuning-1","paper_url":"https://arxiv.org/abs/2407.13690v2","paper_title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","code":"https://github.com/hkust-nlp/dart-math","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":2,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":68,"model":"Claude Instant 1.1 (0-shot chain-of-thought)","metrics":{"Accuracy":"80.9"},"uses_additional_data":false,"paper_date":"2023-07-11","paper":"/paper/model-card-and-evaluations-for-claude-models","paper_url":"https://www-files.anthropic.com/production/images/Model-Card-Claude-2.pdf","paper_title":"Model Card and Evaluations for Claude Models","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":69,"model":"ToRA-Code 34B","metrics":{"Accuracy":"80.7","Parameters (Billion)":"34"},"uses_additional_data":true,"paper_date":"2023-09-29","paper":"/paper/tora-a-tool-integrated-reasoning-agent-for","paper_url":"https://arxiv.org/abs/2309.17452v4","paper_title":"ToRA: A Tool-Integrated Reasoning Agent for Mathematical Problem Solving","code":"https://github.com/microsoft/tora","n_code_links":1,"syntology":{"n_ran":6,"n_unverified":6,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":70,"model":"OpenMath-CodeLlama-34B (w/ code)","metrics":{"Accuracy":"80.7","Parameters (Billion)":"34"},"uses_additional_data":true,"paper_date":"2024-02-15","paper":"/paper/openmathinstruct-1-a-1-8-million-math","paper_url":"https://arxiv.org/abs/2402.10176v2","paper_title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","code":"https://github.com/kipok/nemo-skills","n_code_links":1,"syntology":null},{"rank_in_archive_order":71,"model":"PaLM 2 (few-shot, k=8, CoT)","metrics":{"Accuracy":"80.7"},"uses_additional_data":false,"paper_date":"2023-05-17","paper":"/paper/palm-2-technical-report-1","paper_url":"https://arxiv.org/abs/2305.10403v3","paper_title":"PaLM 2 Technical Report","code":"https://github.com/eternityyw/tram-benchmark","n_code_links":1,"syntology":null},{"rank_in_archive_order":72,"model":"MMOS-DeepSeekMath-7B(0-shot)","metrics":{"Accuracy":"80.5","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2024-02-23","paper":"/paper/an-empirical-study-of-data-ability-boundary","paper_url":"https://arxiv.org/abs/2403.00799v1","paper_title":"An Empirical Study of Data Ability Boundary in LLMs' Math Reasoning","code":"https://github.com/cyzhh/MMOS","n_code_links":1,"syntology":{"n_ran":9,"n_unverified":2,"n_samples":11,"n_pointer_only_licence":11}},{"rank_in_archive_order":73,"model":"MMOS-CODE-34B(0-shot)","metrics":{"Accuracy":"80.4","Parameters (Billion)":"34"},"uses_additional_data":true,"paper_date":"2024-02-23","paper":"/paper/an-empirical-study-of-data-ability-boundary","paper_url":"https://arxiv.org/abs/2403.00799v1","paper_title":"An Empirical Study of Data Ability Boundary in LLMs' Math Reasoning","code":"https://github.com/cyzhh/MMOS","n_code_links":1,"syntology":{"n_ran":9,"n_unverified":2,"n_samples":11,"n_pointer_only_licence":11}},{"rank_in_archive_order":74,"model":"OpenMath-Mistral-7B (w/ code)","metrics":{"Accuracy":"80.2","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2024-02-15","paper":"/paper/openmathinstruct-1-a-1-8-million-math","paper_url":"https://arxiv.org/abs/2402.10176v2","paper_title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","code":"https://github.com/kipok/nemo-skills","n_code_links":1,"syntology":null},{"rank_in_archive_order":75,"model":"Self-Evaluation Guided Decoding (Codex, PAL, single reasoning chain, 9-shot gen, 5-shot eval)","metrics":{"Accuracy":"80.2"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":76,"model":"OpenMath-CodeLlama-13B (w/ code)","metrics":{"Accuracy":"78.8","Parameters (Billion)":"13"},"uses_additional_data":true,"paper_date":"2024-02-15","paper":"/paper/openmathinstruct-1-a-1-8-million-math","paper_url":"https://arxiv.org/abs/2402.10176v2","paper_title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","code":"https://github.com/kipok/nemo-skills","n_code_links":1,"syntology":null},{"rank_in_archive_order":77,"model":"Minerva 540B (CoT)","metrics":{"Accuracy":"78.5","Parameters (Billion)":"540"},"uses_additional_data":false,"paper_date":"2022-06-29","paper":"/paper/solving-quantitative-reasoning-problems-with","paper_url":"https://arxiv.org/abs/2206.14858v2","paper_title":"Solving Quantitative Reasoning Problems with Language Models","code":"https://github.com/gair-nlp/abel","n_code_links":1,"syntology":null},{"rank_in_archive_order":78,"model":"Camelidae-8×34B (5-shot)","metrics":{"Accuracy":"78.3"},"uses_additional_data":false,"paper_date":"2024-01-05","paper":"/paper/parameter-efficient-sparsity-crafting-from","paper_url":"https://arxiv.org/abs/2401.02731v4","paper_title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","code":"https://github.com/wuhy68/parameter-efficient-moe","n_code_links":2,"syntology":null},{"rank_in_archive_order":79,"model":"Qwen2idae-16x14B (5-shot)","metrics":{"Accuracy":"77.8"},"uses_additional_data":false,"paper_date":"2024-01-05","paper":"/paper/parameter-efficient-sparsity-crafting-from","paper_url":"https://arxiv.org/abs/2401.02731v4","paper_title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","code":"https://github.com/wuhy68/parameter-efficient-moe","n_code_links":2,"syntology":null},{"rank_in_archive_order":80,"model":"MetaMath-Mistral-7B","metrics":{"Accuracy":"77.7","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2023-09-21","paper":"/paper/metamath-bootstrap-your-own-mathematical","paper_url":"https://arxiv.org/abs/2309.12284v4","paper_title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","code":"https://github.com/meta-math/MetaMath","n_code_links":1,"syntology":{"n_ran":15,"n_unverified":7,"n_samples":22,"n_pointer_only_licence":0}},{"rank_in_archive_order":81,"model":"OpenChat-3.5 7B","metrics":{"Accuracy":"77.3","Parameters (Billion)":"7"},"uses_additional_data":false,"paper_date":"2023-09-20","paper":"/paper/openchat-advancing-open-source-language","paper_url":"https://arxiv.org/abs/2309.11235v2","paper_title":"OpenChat: Advancing Open-source Language Models with Mixed-Quality Data","code":"https://github.com/imoneoi/openchat","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":1,"n_samples":1,"n_pointer_only_licence":0}},{"rank_in_archive_order":82,"model":"DeepMind 70B Model (STaR, maj1@96)","metrics":{"Accuracy":"76.5","Parameters (Billion)":"70"},"uses_additional_data":true,"paper_date":"2022-11-25","paper":"/paper/solving-math-word-problems-with-process-and","paper_url":"https://arxiv.org/abs/2211.14275v1","paper_title":"Solving math word problems with process- and outcome-based feedback","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":83,"model":"Arithmo2-Mistral-7B","metrics":{"Accuracy":"76.4","Parameters (Billion)":"7"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":84,"model":"OpenMath-CodeLlama-7B (w/ code)","metrics":{"Accuracy":"75.9","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2024-02-15","paper":"/paper/openmathinstruct-1-a-1-8-million-math","paper_url":"https://arxiv.org/abs/2402.10176v2","paper_title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","code":"https://github.com/kipok/nemo-skills","n_code_links":1,"syntology":null},{"rank_in_archive_order":85,"model":"ToRA-Code 13B","metrics":{"Accuracy":"75.8","Parameters (Billion)":"13"},"uses_additional_data":true,"paper_date":"2023-09-29","paper":"/paper/tora-a-tool-integrated-reasoning-agent-for","paper_url":"https://arxiv.org/abs/2309.17452v4","paper_title":"ToRA: A Tool-Integrated Reasoning Agent for Mathematical Problem Solving","code":"https://github.com/microsoft/tora","n_code_links":1,"syntology":{"n_ran":6,"n_unverified":6,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":86,"model":"Arithmo-Mistral-7B","metrics":{"Accuracy":"74.7","Parameters (Billion)":"7"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":87,"model":"PaLM 540B maj1@40 (8-shot)","metrics":{"Accuracy":"74.4","Parameters (Billion)":"540"},"uses_additional_data":true,"paper_date":"2022-03-21","paper":"/paper/self-consistency-improves-chain-of-thought","paper_url":"https://arxiv.org/abs/2203.11171v4","paper_title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","code":"https://github.com/codelion/optillm/blob/main/optillm/self_consistency.py","n_code_links":3,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":0}},{"rank_in_archive_order":88,"model":"PaLM 540B (Self Consistency)","metrics":{"Accuracy":"74.4","Parameters (Billion)":"540"},"uses_additional_data":false,"paper_date":"2022-10-20","paper":"/paper/large-language-models-can-self-improve","paper_url":"https://arxiv.org/abs/2210.11610v2","paper_title":"Large Language Models Can Self-Improve","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":89,"model":"Phi-GSM 2.7B (fine-tuned)","metrics":{"Accuracy":"74.3","Parameters (Billion)":"2.7"},"uses_additional_data":false,"paper_date":"2023-12-14","paper":"/paper/tinygsm-achieving-80-on-gsm8k-with-small","paper_url":"https://arxiv.org/abs/2312.09241v1","paper_title":"TinyGSM: achieving >80% on GSM8k with small language models","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":90,"model":"MathCoder-CL-13B","metrics":{"Accuracy":"74.1","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2023-10-05","paper":"/paper/mathcoder-seamless-code-integration-in-llms","paper_url":"https://arxiv.org/abs/2310.03731v1","paper_title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","code":"https://github.com/mathllm/mathcoder","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":0}},{"rank_in_archive_order":91,"model":"MuggleMATH 13B","metrics":{"Accuracy":"74","Parameters (Billion)":"13"},"uses_additional_data":true,"paper_date":"2023-10-09","paper":"/paper/query-and-response-augmentation-cannot-help","paper_url":"https://arxiv.org/abs/2310.05506v3","paper_title":"MuggleMath: Assessing the Impact of Query and Response Augmentation on Math Reasoning","code":"https://github.com/ofa-sys/gsm8k-screl","n_code_links":1,"syntology":null},{"rank_in_archive_order":92,"model":"MMOS-CODE-7B(0-shot)","metrics":{"Accuracy":"73.9","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2024-02-23","paper":"/paper/an-empirical-study-of-data-ability-boundary","paper_url":"https://arxiv.org/abs/2403.00799v1","paper_title":"An Empirical Study of Data Ability Boundary in LLMs' Math Reasoning","code":"https://github.com/cyzhh/MMOS","n_code_links":1,"syntology":{"n_ran":9,"n_unverified":2,"n_samples":11,"n_pointer_only_licence":11}},{"rank_in_archive_order":93,"model":"CodeT5+","metrics":{"Accuracy":"73.8","Parameters (Billion)":"0.77"},"uses_additional_data":false,"paper_date":"2023-05-13","paper":"/paper/codet5-open-code-large-language-models-for","paper_url":"https://arxiv.org/abs/2305.07922v2","paper_title":"CodeT5+: Open Code Large Language Models for Code Understanding and Generation","code":"https://github.com/salesforce/codet5","n_code_links":2,"syntology":{"n_ran":3,"n_unverified":1,"n_samples":4,"n_pointer_only_licence":0}},{"rank_in_archive_order":94,"model":"Llama-3.3-70B + CAPO","metrics":{"Accuracy":"73.73"},"uses_additional_data":false,"paper_date":"2025-04-22","paper":"/paper/capo-cost-aware-prompt-optimization","paper_url":"https://arxiv.org/abs/2504.16005v3","paper_title":"CAPO: Cost-Aware Prompt Optimization","code":"https://github.com/finitearth/promptolution","n_code_links":2,"syntology":null},{"rank_in_archive_order":95,"model":"OVM-Llama2-7B (verify100@1)","metrics":{"Accuracy":"73.7","Parameters (Billion)":"7"},"uses_additional_data":false,"paper_date":"2023-11-16","paper":"/paper/outcome-supervised-verifiers-for-planning-in","paper_url":"https://arxiv.org/abs/2311.09724v2","paper_title":"OVM, Outcome-supervised Value Models for Planning in Mathematical Reasoning","code":"https://github.com/freedomintelligence/ovm","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":0,"n_samples":3,"n_pointer_only_licence":3}},{"rank_in_archive_order":96,"model":"PaLM 540B (Self Improvement, CoT Prompting)","metrics":{"Accuracy":"73.5","Parameters (Billion)":"540"},"uses_additional_data":false,"paper_date":"2022-10-20","paper":"/paper/large-language-models-can-self-improve","paper_url":"https://arxiv.org/abs/2210.11610v2","paper_title":"Large Language Models Can Self-Improve","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":97,"model":"KwaiYiiMath 13B","metrics":{"Accuracy":"73.3","Parameters (Billion)":"13"},"uses_additional_data":true,"paper_date":"2023-10-11","paper":"/paper/kwaiyiimath-technical-report","paper_url":"https://arxiv.org/abs/2310.07488v2","paper_title":"KwaiYiiMath: Technical Report","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":98,"model":"ToRA-Code 7B","metrics":{"Accuracy":"72.6","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2023-09-29","paper":"/paper/tora-a-tool-integrated-reasoning-agent-for","paper_url":"https://arxiv.org/abs/2309.17452v4","paper_title":"ToRA: A Tool-Integrated Reasoning Agent for Mathematical Problem Solving","code":"https://github.com/microsoft/tora","n_code_links":1,"syntology":{"n_ran":6,"n_unverified":6,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":99,"model":"MathCoder-L-13B","metrics":{"Accuracy":"72.6","Parameters (Billion)":"13"},"uses_additional_data":true,"paper_date":"2023-10-05","paper":"/paper/mathcoder-seamless-code-integration-in-llms","paper_url":"https://arxiv.org/abs/2310.03731v1","paper_title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","code":"https://github.com/mathllm/mathcoder","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":0}},{"rank_in_archive_order":100,"model":"DBRX Base 132B","metrics":{"Accuracy":"72.3"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":101,"model":"Self-Evaluation Guided Decoding (Codex, CoT, single reasoning chain, 9-shot gen, 5-shot eval)","metrics":{"Accuracy":"71.9"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":102,"model":"MetaMath 13B","metrics":{"Accuracy":"71.0","Parameters (Billion)":"13"},"uses_additional_data":true,"paper_date":"2023-09-21","paper":"/paper/metamath-bootstrap-your-own-mathematical","paper_url":"https://arxiv.org/abs/2309.12284v4","paper_title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","code":"https://github.com/meta-math/MetaMath","n_code_links":1,"syntology":{"n_ran":15,"n_unverified":7,"n_samples":22,"n_pointer_only_licence":0}},{"rank_in_archive_order":103,"model":"MuggleMATH 7B","metrics":{"Accuracy":"69.8","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2023-10-09","paper":"/paper/query-and-response-augmentation-cannot-help","paper_url":"https://arxiv.org/abs/2310.05506v3","paper_title":"MuggleMath: Assessing the Impact of Query and Response Augmentation on Math Reasoning","code":"https://github.com/ofa-sys/gsm8k-screl","n_code_links":1,"syntology":null},{"rank_in_archive_order":104,"model":"LLaMA 65B-maj1@k","metrics":{"Accuracy":"69.7","Parameters (Billion)":"65"},"uses_additional_data":false,"paper_date":"2023-02-27","paper":"/paper/llama-open-and-efficient-foundation-language-1","paper_url":"https://arxiv.org/abs/2302.13971v1","paper_title":"LLaMA: Open and Efficient Foundation Language Models","code":"https://github.com/huggingface/transformers","n_code_links":57,"syntology":{"n_ran":37,"n_unverified":21,"n_samples":58,"n_pointer_only_licence":4}},{"rank_in_archive_order":105,"model":"Minerva 62B (maj1@100)","metrics":{"Accuracy":"68.5","Parameters (Billion)":"62"},"uses_additional_data":true,"paper_date":"2022-06-29","paper":"/paper/solving-quantitative-reasoning-problems-with","paper_url":"https://arxiv.org/abs/2206.14858v2","paper_title":"Solving Quantitative Reasoning Problems with Language Models","code":"https://github.com/gair-nlp/abel","n_code_links":1,"syntology":null},{"rank_in_archive_order":106,"model":"code-davinci-002 (Least-to-Most Prompting)","metrics":{"Accuracy":"68.01","Parameters (Billion)":"175"},"uses_additional_data":false,"paper_date":"2022-05-21","paper":"/paper/least-to-most-prompting-enables-complex","paper_url":"https://arxiv.org/abs/2205.10625v3","paper_title":"Least-to-Most Prompting Enables Complex Reasoning in Large Language Models","code":"https://github.com/RUCAIBox/LLMBox","n_code_links":1,"syntology":null},{"rank_in_archive_order":107,"model":"MathCoder-CL-7B","metrics":{"Accuracy":"67.8","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2023-10-05","paper":"/paper/mathcoder-seamless-code-integration-in-llms","paper_url":"https://arxiv.org/abs/2310.03731v1","paper_title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","code":"https://github.com/mathllm/mathcoder","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":0}},{"rank_in_archive_order":108,"model":"DBRX Instruct 132B","metrics":{"Accuracy":"66.9"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":109,"model":"MetaMath 7B","metrics":{"Accuracy":"66.4","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2023-09-21","paper":"/paper/metamath-bootstrap-your-own-mathematical","paper_url":"https://arxiv.org/abs/2309.12284v4","paper_title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","code":"https://github.com/meta-math/MetaMath","n_code_links":1,"syntology":{"n_ran":15,"n_unverified":7,"n_samples":22,"n_pointer_only_licence":0}},{"rank_in_archive_order":110,"model":"Mistral-Small-24B + CAPO","metrics":{"Accuracy":"65.07"},"uses_additional_data":false,"paper_date":"2025-04-22","paper":"/paper/capo-cost-aware-prompt-optimization","paper_url":"https://arxiv.org/abs/2504.16005v3","paper_title":"CAPO: Cost-Aware Prompt Optimization","code":"https://github.com/finitearth/promptolution","n_code_links":2,"syntology":null},{"rank_in_archive_order":111,"model":"RFT 70B","metrics":{"Accuracy":"64.8","Parameters (Billion)":"79"},"uses_additional_data":true,"paper_date":"2023-08-03","paper":"/paper/scaling-relationship-on-learning-mathematical","paper_url":"https://arxiv.org/abs/2308.01825v2","paper_title":"Scaling Relationship on Learning Mathematical Reasoning with Large Language Models","code":"https://github.com/ofa-sys/gsm8k-screl","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":0,"n_samples":3,"n_pointer_only_licence":3}},{"rank_in_archive_order":112,"model":"MathCoder-L-7B","metrics":{"Accuracy":"64.2","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2023-10-05","paper":"/paper/mathcoder-seamless-code-integration-in-llms","paper_url":"https://arxiv.org/abs/2310.03731v1","paper_title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","code":"https://github.com/mathllm/mathcoder","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":0}},{"rank_in_archive_order":113,"model":"WizardMath-13B-V1.0","metrics":{"Accuracy":"63.9","Parameters (Billion)":"13"},"uses_additional_data":true,"paper_date":"2023-08-18","paper":"/paper/wizardmath-empowering-mathematical-reasoning","paper_url":"https://arxiv.org/abs/2308.09583v2","paper_title":"WizardMath: Empowering Mathematical Reasoning for Large Language Models via Reinforced Evol-Instruct","code":"https://github.com/nlpxucan/wizardlm","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":6,"n_samples":16,"n_pointer_only_licence":16}},{"rank_in_archive_order":114,"model":"GPT-J (CoRe)","metrics":{"Accuracy":"63.2","Parameters (Billion)":"12"},"uses_additional_data":false,"paper_date":"2022-10-28","paper":"/paper/solving-math-word-problem-via-cooperative","paper_url":"https://arxiv.org/abs/2210.16257v5","paper_title":"Solving Math Word Problems via Cooperative Reasoning induced Language Models","code":"https://github.com/tianhongzxy/core","n_code_links":1,"syntology":null},{"rank_in_archive_order":115,"model":"Llama-2 70B (on 100 first questions, 4-shot, auto-optimized prompting)","metrics":{"Accuracy":"61","Parameters (Billion)":"70"},"uses_additional_data":false,"paper_date":"2024-02-09","paper":"/paper/the-unreasonable-effectiveness-of-eccentric","paper_url":"https://arxiv.org/abs/2402.10949v2","paper_title":"The Unreasonable Effectiveness of Eccentric Automatic Prompts","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":116,"model":"Qwen2.5-32B + CAPO","metrics":{"Accuracy":"60.2"},"uses_additional_data":false,"paper_date":"2025-04-22","paper":"/paper/capo-cost-aware-prompt-optimization","paper_url":"https://arxiv.org/abs/2504.16005v3","paper_title":"CAPO: Cost-Aware Prompt Optimization","code":"https://github.com/finitearth/promptolution","n_code_links":2,"syntology":null},{"rank_in_archive_order":117,"model":"LLaMA 2 70B (CoT-Influx)","metrics":{"Accuracy":"59.59","Parameters (Billion)":"70"},"uses_additional_data":false,"paper_date":"2023-12-14","paper":"/paper/boosting-llm-reasoning-push-the-limits-of-few","paper_url":"https://arxiv.org/abs/2312.08901v3","paper_title":"Fewer is More: Boosting LLM Reasoning with Reinforced Context Pruning","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":118,"model":"Orca 2 13B","metrics":{"Accuracy":"59.14","Parameters (Billion)":"13"},"uses_additional_data":false,"paper_date":"2023-11-18","paper":"/paper/orca-2-teaching-small-language-models-how-to","paper_url":"https://arxiv.org/abs/2311.11045v2","paper_title":"Orca 2: Teaching Small Language Models How to Reason","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":119,"model":"U-PaLM","metrics":{"Accuracy":"58.5","Parameters (Billion)":"540"},"uses_additional_data":false,"paper_date":"2022-10-20","paper":"/paper/transcending-scaling-laws-with-0-1-extra","paper_url":"https://arxiv.org/abs/2210.11399v2","paper_title":"Transcending Scaling Laws with 0.1% Extra Compute","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":120,"model":"PaLM-540B (few-Shot-cot)","metrics":{"Accuracy":"58.1","Parameters (Billion)":"540"},"uses_additional_data":true,"paper_date":"2022-05-24","paper":"/paper/large-language-models-are-zero-shot-reasoners","paper_url":"https://arxiv.org/abs/2205.11916v4","paper_title":"Large Language Models are Zero-Shot Reasoners","code":"https://github.com/kojima-takeshi188/zero_shot_cot","n_code_links":4,"syntology":{"n_ran":3,"n_unverified":1,"n_samples":4,"n_pointer_only_licence":1}},{"rank_in_archive_order":121,"model":"GPT-3.5 (few-shot, k=5)","metrics":{"Accuracy":"57.1"},"uses_additional_data":false,"paper_date":"2023-03-15","paper":"/paper/gpt-4-technical-report-1","paper_url":"https://arxiv.org/abs/2303.08774v5","paper_title":"GPT-4 Technical Report","code":"https://github.com/openai/evals","n_code_links":11,"syntology":{"n_ran":5,"n_unverified":0,"n_samples":5,"n_pointer_only_licence":1}},{"rank_in_archive_order":122,"model":"Minerva 8B (maj5@100)","metrics":{"Accuracy":"56.8","Parameters (Billion)":"8"},"uses_additional_data":false,"paper_date":"2022-06-29","paper":"/paper/solving-quantitative-reasoning-problems-with","paper_url":"https://arxiv.org/abs/2206.14858v2","paper_title":"Solving Quantitative Reasoning Problems with Language Models","code":"https://github.com/gair-nlp/abel","n_code_links":1,"syntology":null},{"rank_in_archive_order":123,"model":"LLaMA 2 70B (on-shot)","metrics":{"Accuracy":"56.8","Parameters (Billion)":"70"},"uses_additional_data":false,"paper_date":"2023-07-18","paper":"/paper/llama-2-open-foundation-and-fine-tuned-chat","paper_url":"https://arxiv.org/abs/2307.09288v2","paper_title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","code":"https://github.com/facebookresearch/llama","n_code_links":19,"syntology":{"n_ran":31,"n_unverified":21,"n_samples":52,"n_pointer_only_licence":16}},{"rank_in_archive_order":124,"model":"PaLM 540B (8-shot)","metrics":{"Accuracy":"56.5","Parameters (Billion)":"540"},"uses_additional_data":true,"paper_date":"2022-06-29","paper":"/paper/solving-quantitative-reasoning-problems-with","paper_url":"https://arxiv.org/abs/2206.14858v2","paper_title":"Solving Quantitative Reasoning Problems with Language Models","code":"https://github.com/gair-nlp/abel","n_code_links":1,"syntology":null},{"rank_in_archive_order":125,"model":"PaLM 540B (CoT Prompting)","metrics":{"Accuracy":"56.5","Parameters (Billion)":"540"},"uses_additional_data":false,"paper_date":"2022-10-20","paper":"/paper/large-language-models-can-self-improve","paper_url":"https://arxiv.org/abs/2210.11610v2","paper_title":"Large Language Models Can Self-Improve","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":126,"model":"RFT 13B","metrics":{"Accuracy":"55.3","Parameters (Billion)":"13"},"uses_additional_data":true,"paper_date":"2023-08-03","paper":"/paper/scaling-relationship-on-learning-mathematical","paper_url":"https://arxiv.org/abs/2308.01825v2","paper_title":"Scaling Relationship on Learning Mathematical Reasoning with Large Language Models","code":"https://github.com/ofa-sys/gsm8k-screl","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":0,"n_samples":3,"n_pointer_only_licence":3}},{"rank_in_archive_order":127,"model":"Finetuned GPT-3 175B + verifier","metrics":{"Accuracy":"55.0","Parameters (Billion)":"175"},"uses_additional_data":true,"paper_date":"2022-05-24","paper":"/paper/large-language-models-are-zero-shot-reasoners","paper_url":"https://arxiv.org/abs/2205.11916v4","paper_title":"Large Language Models are Zero-Shot Reasoners","code":"https://github.com/kojima-takeshi188/zero_shot_cot","n_code_links":4,"syntology":{"n_ran":3,"n_unverified":1,"n_samples":4,"n_pointer_only_licence":1}},{"rank_in_archive_order":128,"model":"WizardMath-7B-V1.0","metrics":{"Accuracy":"54.9","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2023-08-18","paper":"/paper/wizardmath-empowering-mathematical-reasoning","paper_url":"https://arxiv.org/abs/2308.09583v2","paper_title":"WizardMath: Empowering Mathematical Reasoning for Large Language Models via Reinforced Evol-Instruct","code":"https://github.com/nlpxucan/wizardlm","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":6,"n_samples":16,"n_pointer_only_licence":16}},{"rank_in_archive_order":129,"model":"LLaMA 33B-maj1@k","metrics":{"Accuracy":"53.1","Parameters (Billion)":"33"},"uses_additional_data":false,"paper_date":"2023-02-27","paper":"/paper/llama-open-and-efficient-foundation-language-1","paper_url":"https://arxiv.org/abs/2302.13971v1","paper_title":"LLaMA: Open and Efficient Foundation Language Models","code":"https://github.com/huggingface/transformers","n_code_links":57,"syntology":{"n_ran":37,"n_unverified":21,"n_samples":58,"n_pointer_only_licence":4}},{"rank_in_archive_order":130,"model":"Minerva 62B (8-shot)","metrics":{"Accuracy":" 52.4","Parameters (Billion)":"62"},"uses_additional_data":true,"paper_date":"2022-06-29","paper":"/paper/solving-quantitative-reasoning-problems-with","paper_url":"https://arxiv.org/abs/2206.14858v2","paper_title":"Solving Quantitative Reasoning Problems with Language Models","code":"https://github.com/gair-nlp/abel","n_code_links":1,"syntology":null},{"rank_in_archive_order":131,"model":"Mistral 7B (maj@8)","metrics":{"Accuracy":"52.2","Parameters (Billion)":"7"},"uses_additional_data":false,"paper_date":"2023-10-10","paper":"/paper/mistral-7b","paper_url":"https://arxiv.org/abs/2310.06825v1","paper_title":"Mistral 7B","code":"https://github.com/mistralai/mistral-src","n_code_links":6,"syntology":{"n_ran":10,"n_unverified":1,"n_samples":11,"n_pointer_only_licence":1}},{"rank_in_archive_order":132,"model":"Llemma 34B","metrics":{"Accuracy":"51.5","Parameters (Billion)":"34"},"uses_additional_data":false,"paper_date":"2023-10-16","paper":"/paper/llemma-an-open-language-model-for-mathematics","paper_url":"https://arxiv.org/abs/2310.10631v3","paper_title":"Llemma: An Open Language Model For Mathematics","code":"https://github.com/eleutherai/gpt-neox","n_code_links":4,"syntology":{"n_ran":6,"n_unverified":2,"n_samples":8,"n_pointer_only_licence":0}},{"rank_in_archive_order":133,"model":"Text-davinci-002-175B (zero-plus-few-Shot-cot (8 samples))","metrics":{"Accuracy":"51.5","Parameters (Billion)":"175"},"uses_additional_data":true,"paper_date":"2022-05-24","paper":"/paper/large-language-models-are-zero-shot-reasoners","paper_url":"https://arxiv.org/abs/2205.11916v4","paper_title":"Large Language Models are Zero-Shot Reasoners","code":"https://github.com/kojima-takeshi188/zero_shot_cot","n_code_links":4,"syntology":{"n_ran":3,"n_unverified":1,"n_samples":4,"n_pointer_only_licence":1}},{"rank_in_archive_order":134,"model":"RFT 7B","metrics":{"Accuracy":"51.2","Parameters (Billion)":"7"},"uses_additional_data":true,"paper_date":"2023-08-03","paper":"/paper/scaling-relationship-on-learning-mathematical","paper_url":"https://arxiv.org/abs/2308.01825v2","paper_title":"Scaling Relationship on Learning Mathematical Reasoning with Large Language Models","code":"https://github.com/ofa-sys/gsm8k-screl","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":0,"n_samples":3,"n_pointer_only_licence":3}},{"rank_in_archive_order":135,"model":"LLaMA 65B","metrics":{"Accuracy":"50.9","Parameters (Billion)":"65"},"uses_additional_data":false,"paper_date":"2023-02-27","paper":"/paper/llama-open-and-efficient-foundation-language-1","paper_url":"https://arxiv.org/abs/2302.13971v1","paper_title":"LLaMA: Open and Efficient Foundation Language Models","code":"https://github.com/huggingface/transformers","n_code_links":57,"syntology":{"n_ran":37,"n_unverified":21,"n_samples":58,"n_pointer_only_licence":4}},{"rank_in_archive_order":136,"model":"Orca 2 7B","metrics":{"Accuracy":"47.23","Parameters (Billion)":"7"},"uses_additional_data":false,"paper_date":"2023-11-18","paper":"/paper/orca-2-teaching-small-language-models-how-to","paper_url":"https://arxiv.org/abs/2311.11045v2","paper_title":"Orca 2: Teaching Small Language Models How to Reason","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":137,"model":"Llama-2 13B (on 100 first questions, 4-shot, auto-optimized prompting)","metrics":{"Accuracy":"43","Parameters (Billion)":"13"},"uses_additional_data":false,"paper_date":"2024-02-09","paper":"/paper/the-unreasonable-effectiveness-of-eccentric","paper_url":"https://arxiv.org/abs/2402.10949v2","paper_title":"The Unreasonable Effectiveness of Eccentric Automatic Prompts","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":138,"model":"text-davinci-002 175B (2-shot, CoT)","metrics":{"Accuracy":"41.3","Parameters (Billion)":"175"},"uses_additional_data":true,"paper_date":"2022-05-24","paper":"/paper/large-language-models-are-zero-shot-reasoners","paper_url":"https://arxiv.org/abs/2205.11916v4","paper_title":"Large Language Models are Zero-Shot Reasoners","code":"https://github.com/kojima-takeshi188/zero_shot_cot","n_code_links":4,"syntology":{"n_ran":3,"n_unverified":1,"n_samples":4,"n_pointer_only_licence":1}},{"rank_in_archive_order":139,"model":"Mistral 7B (on 100 first questions, 4-shot, auto-optimized prompting)","metrics":{"Accuracy":"41","Parameters (Billion)":"7"},"uses_additional_data":false,"paper_date":"2024-02-09","paper":"/paper/the-unreasonable-effectiveness-of-eccentric","paper_url":"https://arxiv.org/abs/2402.10949v2","paper_title":"The Unreasonable Effectiveness of Eccentric Automatic Prompts","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":140,"model":"text-davinci-002 175B (0-shot, CoT)","metrics":{"Accuracy":"40.7","Parameters (Billion)":"175"},"uses_additional_data":true,"paper_date":"2022-05-24","paper":"/paper/large-language-models-are-zero-shot-reasoners","paper_url":"https://arxiv.org/abs/2205.11916v4","paper_title":"Large Language Models are Zero-Shot Reasoners","code":"https://github.com/kojima-takeshi188/zero_shot_cot","n_code_links":4,"syntology":{"n_ran":3,"n_unverified":1,"n_samples":4,"n_pointer_only_licence":1}},{"rank_in_archive_order":141,"model":"Branch-Train-MiX 4x7B (sampling top-2 experts)","metrics":{"Accuracy":"37.1"},"uses_additional_data":false,"paper_date":"2024-03-12","paper":"/paper/branch-train-mix-mixing-expert-llms-into-a","paper_url":"https://arxiv.org/abs/2403.07816v1","paper_title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","code":"https://github.com/Leeroo-AI/mergoo","n_code_links":1,"syntology":null},{"rank_in_archive_order":142,"model":"Llemma 7B","metrics":{"Accuracy":"36.4","Parameters (Billion)":"7"},"uses_additional_data":false,"paper_date":"2023-10-16","paper":"/paper/llemma-an-open-language-model-for-mathematics","paper_url":"https://arxiv.org/abs/2310.10631v3","paper_title":"Llemma: An Open Language Model For Mathematics","code":"https://github.com/eleutherai/gpt-neox","n_code_links":4,"syntology":{"n_ran":6,"n_unverified":2,"n_samples":8,"n_pointer_only_licence":0}},{"rank_in_archive_order":143,"model":"LLaMA 33B","metrics":{"Accuracy":"35.6","Parameters (Billion)":"33"},"uses_additional_data":false,"paper_date":"2023-02-27","paper":"/paper/llama-open-and-efficient-foundation-language-1","paper_url":"https://arxiv.org/abs/2302.13971v1","paper_title":"LLaMA: Open and Efficient Foundation Language Models","code":"https://github.com/huggingface/transformers","n_code_links":57,"syntology":{"n_ran":37,"n_unverified":21,"n_samples":58,"n_pointer_only_licence":4}},{"rank_in_archive_order":144,"model":"Vicuna (SYRELM)","metrics":{"Accuracy":"35.2","Parameters (Billion)":"13"},"uses_additional_data":true,"paper_date":"2023-12-09","paper":"/paper/frugal-lms-trained-to-invoke-symbolic-solvers","paper_url":"https://arxiv.org/abs/2312.05571v2","paper_title":"Frugal LMs Trained to Invoke Symbolic Solvers Achieve Parameter-Efficient Arithmetic Reasoning","code":"https://github.com/joykirat18/syrelm","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":2}},{"rank_in_archive_order":145,"model":"PaLM 62B (8-shot)","metrics":{"Accuracy":"33.0","Parameters (Billion)":"62"},"uses_additional_data":true,"paper_date":"2022-06-29","paper":"/paper/solving-quantitative-reasoning-problems-with","paper_url":"https://arxiv.org/abs/2206.14858v2","paper_title":"Solving Quantitative Reasoning Problems with Language Models","code":"https://github.com/gair-nlp/abel","n_code_links":1,"syntology":null},{"rank_in_archive_order":146,"model":"PaLM 540B (Self Improvement, Standard-Prompting)","metrics":{"Accuracy":"32.2","Parameters (Billion)":"540"},"uses_additional_data":false,"paper_date":"2022-10-20","paper":"/paper/large-language-models-can-self-improve","paper_url":"https://arxiv.org/abs/2210.11610v2","paper_title":"Large Language Models Can Self-Improve","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":147,"model":"LLaMA 13B-maj1@k","metrics":{"Accuracy":"29.3","Parameters (Billion)":"13"},"uses_additional_data":false,"paper_date":"2023-02-27","paper":"/paper/llama-open-and-efficient-foundation-language-1","paper_url":"https://arxiv.org/abs/2302.13971v1","paper_title":"LLaMA: Open and Efficient Foundation Language Models","code":"https://github.com/huggingface/transformers","n_code_links":57,"syntology":{"n_ran":37,"n_unverified":21,"n_samples":58,"n_pointer_only_licence":4}},{"rank_in_archive_order":148,"model":"Minerva 8B-maj1@k (8-shot)","metrics":{"Accuracy":" 28.4","Parameters (Billion)":"8"},"uses_additional_data":true,"paper_date":"2022-06-29","paper":"/paper/solving-quantitative-reasoning-problems-with","paper_url":"https://arxiv.org/abs/2206.14858v2","paper_title":"Solving Quantitative Reasoning Problems with Language Models","code":"https://github.com/gair-nlp/abel","n_code_links":1,"syntology":null},{"rank_in_archive_order":149,"model":"GPT-2-Medium 355M + question-solution classifier (BS=5)","metrics":{"Accuracy":"20.8","Parameters (Billion)":"0.355"},"uses_additional_data":false,"paper_date":"2022-10-20","paper":"/paper/composing-ensembles-of-pre-trained-models-via","paper_url":"https://arxiv.org/abs/2210.11522v1","paper_title":"Composing Ensembles of Pre-trained Models via Iterative Consensus","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":150,"model":"GPT-Neo-2.7B + Self-Sampling","metrics":{"Accuracy":"19.5","Parameters (Billion)":"2.7"},"uses_additional_data":false,"paper_date":"2022-05-28","paper":"/paper/learning-from-self-sampled-correct-and","paper_url":"https://arxiv.org/abs/2205.14318v2","paper_title":"Learning Math Reasoning from Self-Sampled Correct and Partially-Correct Solutions","code":"https://github.com/microsoft/tracecodegen","n_code_links":1,"syntology":{"n_ran":7,"n_unverified":4,"n_samples":11,"n_pointer_only_licence":0}},{"rank_in_archive_order":151,"model":"GPT-2-Medium 355M (fine-tuned, BS=5)","metrics":{"Accuracy":"18.3","Parameters (Billion)":"0.355"},"uses_additional_data":false,"paper_date":"2022-10-20","paper":"/paper/composing-ensembles-of-pre-trained-models-via","paper_url":"https://arxiv.org/abs/2210.11522v1","paper_title":"Composing Ensembles of Pre-trained Models via Iterative Consensus","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":152,"model":"LLaMA 7B (maj1@k)","metrics":{"Accuracy":"18.1","Parameters (Billion)":"7"},"uses_additional_data":false,"paper_date":"2023-02-27","paper":"/paper/llama-open-and-efficient-foundation-language-1","paper_url":"https://arxiv.org/abs/2302.13971v1","paper_title":"LLaMA: Open and Efficient Foundation Language Models","code":"https://github.com/huggingface/transformers","n_code_links":57,"syntology":{"n_ran":37,"n_unverified":21,"n_samples":58,"n_pointer_only_licence":4}},{"rank_in_archive_order":153,"model":"PaLM 540B (few-shot)","metrics":{"Accuracy":"17.9","Parameters (Billion)":"540"},"uses_additional_data":true,"paper_date":"2022-05-24","paper":"/paper/large-language-models-are-zero-shot-reasoners","paper_url":"https://arxiv.org/abs/2205.11916v4","paper_title":"Large Language Models are Zero-Shot Reasoners","code":"https://github.com/kojima-takeshi188/zero_shot_cot","n_code_links":4,"syntology":{"n_ran":3,"n_unverified":1,"n_samples":4,"n_pointer_only_licence":1}},{"rank_in_archive_order":154,"model":"PaLM 540B (Standard-Prompting)","metrics":{"Accuracy":"17.9","Parameters (Billion)":"540"},"uses_additional_data":false,"paper_date":"2022-10-20","paper":"/paper/large-language-models-can-self-improve","paper_url":"https://arxiv.org/abs/2210.11610v2","paper_title":"Large Language Models Can Self-Improve","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":155,"model":"LLaMA 13B","metrics":{"Accuracy":"17.8","Parameters (Billion)":"13"},"uses_additional_data":false,"paper_date":"2023-02-27","paper":"/paper/llama-open-and-efficient-foundation-language-1","paper_url":"https://arxiv.org/abs/2302.13971v1","paper_title":"LLaMA: Open and Efficient Foundation Language Models","code":"https://github.com/huggingface/transformers","n_code_links":57,"syntology":{"n_ran":37,"n_unverified":21,"n_samples":58,"n_pointer_only_licence":4}},{"rank_in_archive_order":156,"model":"GPT-2-Medium 355M + question-solution classifier (BS=1)","metrics":{"Accuracy":"16.8","Parameters (Billion)":"0.355"},"uses_additional_data":false,"paper_date":"2022-10-20","paper":"/paper/composing-ensembles-of-pre-trained-models-via","paper_url":"https://arxiv.org/abs/2210.11522v1","paper_title":"Composing Ensembles of Pre-trained Models via Iterative Consensus","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":157,"model":"Minerva 8B (8-shot)","metrics":{"Accuracy":"16.2","Parameters (Billion)":"8"},"uses_additional_data":true,"paper_date":"2022-06-29","paper":"/paper/solving-quantitative-reasoning-problems-with","paper_url":"https://arxiv.org/abs/2206.14858v2","paper_title":"Solving Quantitative Reasoning Problems with Language Models","code":"https://github.com/gair-nlp/abel","n_code_links":1,"syntology":null},{"rank_in_archive_order":158,"model":"GPT-2-Medium 355M (BS=5)","metrics":{"Accuracy":"12.2","Parameters (Billion)":"0.355"},"uses_additional_data":false,"paper_date":"2022-10-20","paper":"/paper/composing-ensembles-of-pre-trained-models-via","paper_url":"https://arxiv.org/abs/2210.11522v1","paper_title":"Composing Ensembles of Pre-trained Models via Iterative Consensus","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":159,"model":"LLaMA 7B","metrics":{"Accuracy":"11.0","Parameters (Billion)":"7"},"uses_additional_data":false,"paper_date":"2023-02-27","paper":"/paper/llama-open-and-efficient-foundation-language-1","paper_url":"https://arxiv.org/abs/2302.13971v1","paper_title":"LLaMA: Open and Efficient Foundation Language Models","code":"https://github.com/huggingface/transformers","n_code_links":57,"syntology":{"n_ran":37,"n_unverified":21,"n_samples":58,"n_pointer_only_licence":4}},{"rank_in_archive_order":160,"model":"Text-davinci-002-175B (0-shot)","metrics":{"Accuracy":"10.4","Parameters (Billion)":"175"},"uses_additional_data":true,"paper_date":"2022-05-24","paper":"/paper/large-language-models-are-zero-shot-reasoners","paper_url":"https://arxiv.org/abs/2205.11916v4","paper_title":"Large Language Models are Zero-Shot Reasoners","code":"https://github.com/kojima-takeshi188/zero_shot_cot","n_code_links":4,"syntology":{"n_ran":3,"n_unverified":1,"n_samples":4,"n_pointer_only_licence":1}},{"rank_in_archive_order":161,"model":"GPT-Neo 125M + Self-Sampling","metrics":{"Accuracy":"7.5","Parameters (Billion)":"0.125"},"uses_additional_data":false,"paper_date":"2022-05-28","paper":"/paper/learning-from-self-sampled-correct-and","paper_url":"https://arxiv.org/abs/2205.14318v2","paper_title":"Learning Math Reasoning from Self-Sampled Correct and Partially-Correct Solutions","code":"https://github.com/microsoft/tracecodegen","n_code_links":1,"syntology":{"n_ran":7,"n_unverified":4,"n_samples":11,"n_pointer_only_licence":0}},{"rank_in_archive_order":162,"model":"UL2 20B (chain-of-thought)","metrics":{"Accuracy":"4.4","Parameters (Billion)":"20"},"uses_additional_data":false,"paper_date":"2022-05-10","paper":"/paper/unifying-language-learning-paradigms","paper_url":"https://arxiv.org/abs/2205.05131v3","paper_title":"UL2: Unifying Language Learning Paradigms","code":"https://github.com/google-research/google-research","n_code_links":2,"syntology":{"n_ran":15,"n_unverified":1,"n_samples":16,"n_pointer_only_licence":0}},{"rank_in_archive_order":163,"model":"PaLM 8B (8-shot)","metrics":{"Accuracy":"4.1","Parameters (Billion)":"8"},"uses_additional_data":true,"paper_date":"2022-06-29","paper":"/paper/solving-quantitative-reasoning-problems-with","paper_url":"https://arxiv.org/abs/2206.14858v2","paper_title":"Solving Quantitative Reasoning Problems with Language Models","code":"https://github.com/gair-nlp/abel","n_code_links":1,"syntology":null},{"rank_in_archive_order":164,"model":"UL2 20B (0-shot)","metrics":{"Accuracy":"4.1","Parameters (Billion)":"20"},"uses_additional_data":false,"paper_date":"2022-05-10","paper":"/paper/unifying-language-learning-paradigms","paper_url":"https://arxiv.org/abs/2205.05131v3","paper_title":"UL2: Unifying Language Learning Paradigms","code":"https://github.com/google-research/google-research","n_code_links":2,"syntology":{"n_ran":15,"n_unverified":1,"n_samples":16,"n_pointer_only_licence":0}}],"since_archive":{"claim":"Results that newer papers report for their own method, placed here by Syntology. A model pointed at the cell in the paper's own table; the number was read from that cell and checked against this leaderboard's metric, dataset, split and scale; an independent check that saw this leaderboard's other rows and every other leaderboard on the same dataset accepted it. Not reviewed by the paper's authors or by the archive's editors, and not ranked against the archive rows.","extraction_file_present":true,"measurement":{"test_papers":883,"papers_with_output":881,"judged_true":108,"judged":110,"wilson95_lower":0.9361,"measured_on":"2026-09-24","frozen_commit":"0e3de0df94"},"measurement_note":"blind adjudication of accepted entries on a held-out split of archive papers, rules frozen before the test","coverage":{"sentence":"Syntology has checked 6,885 of the 9,623 papers on this site that are newer than the archive; results from the others appear after they are checked.","complete":false,"papers_newer_than_archive":9623,"papers_checked":6885,"papers_extracted_not_yet_verified":0,"boards_without_verdict":2,"papers_not_yet_extracted":2737},"order":"newest first by month (arXiv date, else the arXiv-id month), then arXiv id descending","columns":[],"entries":[]},"syntology":{"read_at":"2026-09-25T09:33:49+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":77,"rows_with_any_sample_ran":76,"distinct_papers_with_graph_line":28,"distinct_papers_with_any_sample_ran":27,"samples_over_distinct_papers":{"n_ran":259,"n_unverified":111,"n_samples":370,"n_pointer_only_licence":96,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":824,"n_unverified":369,"n_samples":1193,"n_pointer_only_licence":223,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}