{"url":"/task/arithmetic-reasoning","name":"Arithmetic Reasoning","slug":"arithmetic-reasoning","description_markdown":null,"categories":[{"name":"Reasoning","url":"/area/reasoning"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":175,"papers_with_code":112,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":8,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/arithmetic-reasoning-on-gsm8k","slug":"arithmetic-reasoning-on-gsm8k","dataset":"GSM8K","dataset_url":"/dataset/gsm8k","rows_in_archive":164,"metrics":["Accuracy","Parameters (Billion)"],"first_row_in_archive_order":{"model":"Claude 3.5 Sonnet (HPT)","paper_title":"Hierarchical Prompting Taxonomy: A Universal Evaluation Framework for Large Language Models Aligned with Human Cognitive Principles","paper_url":"/paper/hierarchical-prompting-taxonomy-a-universal","paper_date":"2024-06-18","arxiv_id":"2406.12644","code_links":[{"title":"devichand579/HPT","url":"https://github.com/devichand579/HPT"}],"syntology":null}},{"leaderboard":"/sota/arithmetic-reasoning-on-multiarith","slug":"arithmetic-reasoning-on-multiarith","dataset":"MultiArith","dataset_url":null,"rows_in_archive":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Text-davinci-002 (175B)(zero-shot-cot)","paper_title":"Large Language Models are Zero-Shot Reasoners","paper_url":"/paper/large-language-models-are-zero-shot-reasoners","paper_date":"2022-05-24","arxiv_id":"2205.11916","code_links":[{"title":"kojima-takeshi188/zero_shot_cot","url":"https://github.com/kojima-takeshi188/zero_shot_cot"},{"title":"skytliang/multi-agents-debate","url":"https://github.com/skytliang/multi-agents-debate"},{"title":"zongqianwu/st-cot","url":"https://github.com/zongqianwu/st-cot"},{"title":"nicolay-r/reasoning-for-sentiment-analysis-framework","url":"https://github.com/nicolay-r/reasoning-for-sentiment-analysis-framework"}],"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":1}}},{"leaderboard":"/sota/arithmetic-reasoning-on-game-of-24","slug":"arithmetic-reasoning-on-game-of-24","dataset":"Game of 24","dataset_url":"/dataset/game-of-24","rows_in_archive":1,"metrics":["Success"],"first_row_in_archive_order":{"model":"Tree of Thoughts (b=5)","paper_title":"Tree of Thoughts: Deliberate Problem Solving with Large Language Models","paper_url":"/paper/tree-of-thoughts-deliberate-problem-solving-1","paper_date":"2023-05-17","arxiv_id":"2305.10601","code_links":[{"title":"ysymyth/tree-of-thought-llm","url":"https://github.com/ysymyth/tree-of-thought-llm"},{"title":"princeton-nlp/tree-of-thought-llm","url":"https://github.com/princeton-nlp/tree-of-thought-llm"},{"title":"codelion/optillm","url":"https://github.com/codelion/optillm"},{"title":"appl-team/appl","url":"https://github.com/appl-team/appl"},{"title":"huiwy/reflection-on-trees","url":"https://github.com/huiwy/reflection-on-trees"},{"title":"katamb/tree-of-thought-llm-ca","url":"https://github.com/katamb/tree-of-thought-llm-ca"}],"syntology":{"n":24,"n_ran":7,"n_unverified":17,"n_pointer_only":0}}},{"leaderboard":"/sota/arithmetic-reasoning-on-mathmc","slug":"arithmetic-reasoning-on-mathmc","dataset":"MathMC","dataset_url":"/dataset/mathmc","rows_in_archive":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GPT-4 (Teaching-Inspired)","paper_title":"Teaching-Inspired Integrated Prompting Framework: A Novel Approach for Enhancing Reasoning in Large Language Models","paper_url":"/paper/teaching-inspired-integrated-prompting","paper_date":"2024-10-10","arxiv_id":"2410.08068","code_links":[{"title":"sallytan13/teaching-inspired-prompting","url":"https://github.com/sallytan13/teaching-inspired-prompting"}],"syntology":{"n":8,"n_ran":8,"n_unverified":0,"n_pointer_only":8}}},{"leaderboard":"/sota/arithmetic-reasoning-on-mathtof","slug":"arithmetic-reasoning-on-mathtof","dataset":"MathToF","dataset_url":"/dataset/mathtof","rows_in_archive":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GPT-4 (Teaching-Inspired)","paper_title":"Teaching-Inspired Integrated Prompting Framework: A Novel Approach for Enhancing Reasoning in Large Language Models","paper_url":"/paper/teaching-inspired-integrated-prompting","paper_date":"2024-10-10","arxiv_id":"2410.08068","code_links":[{"title":"sallytan13/teaching-inspired-prompting","url":"https://github.com/sallytan13/teaching-inspired-prompting"}],"syntology":{"n":8,"n_ran":8,"n_unverified":0,"n_pointer_only":8}}}],"datasets":[{"url":"/dataset/gsm8k","name":"GSM8K","full_name":"","num_papers_in_archive":1881},{"url":"/dataset/mgsm","name":"MGSM","full_name":"Multilingual Grade School Math","num_papers_in_archive":107},{"url":"/dataset/game-of-24","name":"Game of 24","full_name":"","num_papers_in_archive":61},{"url":"/dataset/gsm-plus","name":"GSM-Plus","full_name":"","num_papers_in_archive":17},{"url":"/dataset/smart-101","name":"SMART-101","full_name":"Simple Multimodal Algorithmic Reasoning Task Dataset","num_papers_in_archive":6},{"url":"/dataset/mathmc","name":"MathMC","full_name":"","num_papers_in_archive":1},{"url":"/dataset/mathtof","name":"MathToF","full_name":"","num_papers_in_archive":1},{"url":"/dataset/polymath","name":"PolyMATH","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":112,"tagged_in_all":175,"items":[{"url":"/paper/llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","arxiv_id":"2302.13971","repositories_listed":57,"syntology":{"n":58,"n_ran":26,"n_unverified":32,"n_pointer_only":4}},{"url":"/paper/llama-2-open-foundation-and-fine-tuned-chat","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","date":"2023-07-18","arxiv_id":"2307.09288","repositories_listed":19,"syntology":{"n":52,"n_ran":31,"n_unverified":21,"n_pointer_only":16}},{"url":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","arxiv_id":"2303.08774","repositories_listed":11,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/qwen2-technical-report","title":"Qwen2 Technical Report","date":"2024-07-15","arxiv_id":"2407.10671","repositories_listed":6,"syntology":null},{"url":"/paper/mistral-7b","title":"Mistral 7B","date":"2023-10-10","arxiv_id":"2310.06825","repositories_listed":6,"syntology":{"n":11,"n_ran":9,"n_unverified":2,"n_pointer_only":1}},{"url":"/paper/tree-of-thoughts-deliberate-problem-solving-1","title":"Tree of Thoughts: Deliberate Problem Solving with Large Language Models","date":"2023-05-17","arxiv_id":"2305.10601","repositories_listed":6,"syntology":{"n":24,"n_ran":7,"n_unverified":17,"n_pointer_only":0}},{"url":"/paper/deepseekmath-pushing-the-limits-of","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","date":"2024-02-05","arxiv_id":"2402.03300","repositories_listed":5,"syntology":{"n":24,"n_ran":8,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/llemma-an-open-language-model-for-mathematics","title":"Llemma: An Open Language Model For Mathematics","date":"2023-10-16","arxiv_id":"2310.10631","repositories_listed":4,"syntology":{"n":8,"n_ran":6,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/large-language-models-are-zero-shot-reasoners","title":"Large Language Models are Zero-Shot Reasoners","date":"2022-05-24","arxiv_id":"2205.11916","repositories_listed":4,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/math-shepherd-a-label-free-step-by-step","title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","date":"2023-12-14","arxiv_id":"2312.08935","repositories_listed":3,"syntology":null},{"url":"/paper/neural-comprehension-language-models-with","title":"Mastering Symbolic Operations: Augmenting Language Models with Compiled Neural Networks","date":"2023-04-04","arxiv_id":"2304.01665","repositories_listed":3,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":1}},{"url":"/paper/sparks-of-artificial-general-intelligence","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","date":"2023-03-22","arxiv_id":"2303.12712","repositories_listed":3,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/pal-program-aided-language-models","title":"PAL: Program-aided Language Models","date":"2022-11-18","arxiv_id":"2211.10435","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/self-consistency-improves-chain-of-thought","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","date":"2022-03-21","arxiv_id":"2203.11171","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/capo-cost-aware-prompt-optimization","title":"CAPO: Cost-Aware Prompt Optimization","date":"2025-04-22","arxiv_id":"2504.16005","repositories_listed":2,"syntology":null},{"url":"/paper/fed-sb-a-silver-bullet-for-extreme","title":"Fed-SB: A Silver Bullet for Extreme Communication Efficiency and Performance in (Private) Federated LoRA Fine-Tuning","date":"2025-02-21","arxiv_id":"2502.15436","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":4}},{"url":"/paper/exact-aggregation-for-federated-and-efficient","title":"FedEx-LoRA: Exact Aggregation for Federated and Efficient Fine-Tuning of Foundation Models","date":"2024-10-12","arxiv_id":"2410.09432","repositories_listed":2,"syntology":null},{"url":"/paper/an-investigation-of-neuron-activation-as-a","title":"An Investigation of Neuron Activation as a Unified Lens to Explain Chain-of-Thought Eliciting Arithmetic Reasoning of LLMs","date":"2024-06-18","arxiv_id":"2406.12288","repositories_listed":2,"syntology":{"n":36,"n_ran":33,"n_unverified":3,"n_pointer_only":11}},{"url":"/paper/buffer-of-thoughts-thought-augmented","title":"Buffer of Thoughts: Thought-Augmented Reasoning with Large Language Models","date":"2024-06-06","arxiv_id":"2406.04271","repositories_listed":2,"syntology":{"n":6,"n_ran":1,"n_unverified":5,"n_pointer_only":1}},{"url":"/paper/reft-representation-finetuning-for-language","title":"ReFT: Representation Finetuning for Language Models","date":"2024-04-04","arxiv_id":"2404.03592","repositories_listed":2,"syntology":{"n":14,"n_ran":1,"n_unverified":13,"n_pointer_only":0}},{"url":"/paper/parameter-efficient-sparsity-crafting-from","title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","date":"2024-01-05","arxiv_id":"2401.02731","repositories_listed":2,"syntology":null},{"url":"/paper/offline-prompt-evaluation-and-optimization","title":"Query-Dependent Prompt Evaluation and Optimization with Offline Inverse RL","date":"2023-09-13","arxiv_id":"2309.06553","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/codet5-open-code-large-language-models-for","title":"CodeT5+: Open Code Large Language Models for Code Understanding and Generation","date":"2023-05-13","arxiv_id":"2305.07922","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/llm-adapters-an-adapter-family-for-parameter","title":"LLM-Adapters: An Adapter Family for Parameter-Efficient Fine-Tuning of Large Language Models","date":"2023-04-04","arxiv_id":"2304.01933","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/automatic-prompt-augmentation-and-selection","title":"Automatic Prompt Augmentation and Selection with Chain-of-Thought from Labeled Data","date":"2023-02-24","arxiv_id":"2302.12822","repositories_listed":2,"syntology":null},{"url":"/paper/batch-prompting-efficient-inference-with","title":"Batch Prompting: Efficient Inference with Large Language Model APIs","date":"2023-01-19","arxiv_id":"2301.08721","repositories_listed":2,"syntology":null},{"url":"/paper/reasoning-with-language-model-prompting-a","title":"Reasoning with Language Model Prompting: A Survey","date":"2022-12-19","arxiv_id":"2212.09597","repositories_listed":2,"syntology":null},{"url":"/paper/unifying-language-learning-paradigms","title":"UL2: Unifying Language Learning Paradigms","date":"2022-05-10","arxiv_id":"2205.05131","repositories_listed":2,"syntology":{"n":16,"n_ran":0,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/learning-to-reason-for-text-generation-from","title":"Learning to Reason for Text Generation from Scientific Tables","date":"2021-04-16","arxiv_id":"2104.08296","repositories_listed":2,"syntology":null},{"url":"/paper/dcr-quantifying-data-contamination-in-llms","title":"DCR: Quantifying Data Contamination in LLMs Evaluation","date":"2025-07-15","arxiv_id":"2507.11405","repositories_listed":1,"syntology":null}],"syntology_records":20,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}