{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/gsm8k/papers/5","list_of":"/task/gsm8k","task":"GSM8K","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":5,"rows_per_page":100,"rows":[401,439],"of":439,"counts":{"archive_papers_tagged":439,"with_a_code_link":209,"where_syntology_ran_a_sample":116,"not_listed_spam_title":0,"listed":439,"listed_where_code_ran":116,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":96,"every_run_a_failure_of_syntologys_instrument":20,"listed_with_a_run_with_no_instrument_failure":96,"listed_every_run_a_failure_of_syntologys_instrument":20,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/gsm8k","prev":"/task/gsm8k/papers/4","next":null,"papers":[{"url":null,"slug":"multi-step-problem-solving-through-a-verifier","title":"Multi-step Problem Solving Through a Verifier: An Empirical Analysis on Model-induced Process Supervision","date":"2024-02-05","arxiv_id":"2402.02658","repositories_listed":0,"syntology":null},{"url":null,"slug":"yoda-teacher-student-progressive-learning-for","title":"YODA: Teacher-Student Progressive Learning for Language Models","date":"2024-01-28","arxiv_id":"2401.15670","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-imagine-effective-unimodal-reasoning","title":"Self-Imagine: Effective Unimodal Reasoning with Multimodal Models using Self-Imagination","date":"2024-01-16","arxiv_id":"2401.08025","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-the-impact-of-prompting-persona-and","title":"Assessing the Impact of Prompting Methods on ChatGPT's Mathematical Capabilities","date":"2023-12-22","arxiv_id":"2312.15006","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-good-to-great-improving-math-reasoning","title":"From Good to Great: Improving Math Reasoning with Tool-Augmented Interleaf Prompting","date":"2023-12-18","arxiv_id":"2401.05384","repositories_listed":0,"syntology":null},{"url":"/paper/boosting-llm-reasoning-push-the-limits-of-few","slug":"boosting-llm-reasoning-push-the-limits-of-few","title":"Fewer is More: Boosting LLM Reasoning with Reinforced Context Pruning","date":"2023-12-14","arxiv_id":"2312.08901","repositories_listed":0,"syntology":null},{"url":"/paper/tinygsm-achieving-80-on-gsm8k-with-small","slug":"tinygsm-achieving-80-on-gsm8k-with-small","title":"TinyGSM: achieving >80% on GSM8k with small language models","date":"2023-12-14","arxiv_id":"2312.09241","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-chain-of-thought-via-latent-variable-1","title":"Training Chain-of-Thought via Latent-Variable Inference","date":"2023-11-28","arxiv_id":"2312.02179","repositories_listed":0,"syntology":null},{"url":null,"slug":"first-step-advantage-importance-of-starting","title":"First-Step Advantage: Importance of Starting Right in Multi-Step Math Reasoning","date":"2023-11-14","arxiv_id":"2311.07945","repositories_listed":0,"syntology":null},{"url":null,"slug":"saie-framework-support-alone-isn-t-enough","title":"SAIE Framework: Support Alone Isn't Enough -- Advancing LLM Training with Adversarial Remarks","date":"2023-11-14","arxiv_id":"2311.08107","repositories_listed":0,"syntology":null},{"url":"/paper/the-art-of-llm-refinement-ask-refine-and","slug":"the-art-of-llm-refinement-ask-refine-and","title":"The ART of LLM Refinement: Ask, Refine, and Trust","date":"2023-11-14","arxiv_id":"2311.07961","repositories_listed":0,"syntology":null},{"url":null,"slug":"let-s-reinforce-step-by-step","title":"Let's Reinforce Step by Step","date":"2023-11-10","arxiv_id":"2311.05821","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-engineering-a-prompt-engineer","title":"Prompt Engineering a Prompt Engineer","date":"2023-11-09","arxiv_id":"2311.05661","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-alignment-ceiling-objective-mismatch-in","title":"The Alignment Ceiling: Objective Mismatch in Reinforcement Learning from Human Feedback","date":"2023-10-31","arxiv_id":"2311.00168","repositories_listed":0,"syntology":null},{"url":null,"slug":"sego-sequential-subgoal-optimization-for","title":"SEGO: Sequential Subgoal Optimization for Mathematical Problem-Solving","date":"2023-10-19","arxiv_id":"2310.12960","repositories_listed":0,"syntology":null},{"url":null,"slug":"let-s-reward-step-by-step-step-level-reward","title":"Let's reward step by step: Step-Level reward model as the Navigators for Reasoning","date":"2023-10-16","arxiv_id":"2310.10080","repositories_listed":0,"syntology":null},{"url":null,"slug":"lobass-gauging-learnability-in-supervised","title":"DavIR: Data Selection via Implicit Reward for Large Language Models","date":"2023-10-16","arxiv_id":"2310.13008","repositories_listed":0,"syntology":null},{"url":"/paper/kwaiyiimath-technical-report","slug":"kwaiyiimath-technical-report","title":"KwaiYiiMath: Technical Report","date":"2023-10-11","arxiv_id":"2310.07488","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-words-to-watts-benchmarking-the-energy","title":"From Words to Watts: Benchmarking the Energy Costs of Large Language Model Inference","date":"2023-10-04","arxiv_id":"2310.03003","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-as-analogical-reasoners","title":"Large Language Models as Analogical Reasoners","date":"2023-10-03","arxiv_id":"2310.01714","repositories_listed":0,"syntology":null},{"url":null,"slug":"think-before-you-speak-training-language","title":"Think before you speak: Training Language Models With Pause Tokens","date":"2023-10-03","arxiv_id":"2310.02226","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-llm-agents-through-communication","title":"Adapting LLM Agents with Universal Feedback in Communication","date":"2023-10-01","arxiv_id":"2310.01444","repositories_listed":0,"syntology":null},{"url":null,"slug":"upar-a-kantian-inspired-prompting-framework","title":"UPAR: A Kantian-Inspired Prompting Framework for Enhancing Large Language Model Capabilities","date":"2023-09-30","arxiv_id":"2310.01441","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-decoding-improves-reasoning-in","title":"Contrastive Decoding Improves Reasoning in Large Language Models","date":"2023-09-17","arxiv_id":"2309.09117","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-an-lm-to-generate-prolog-predicates","title":"Exploring an LM to generate Prolog Predicates from Mathematics Questions","date":"2023-09-07","arxiv_id":"2309.03667","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathattack-attacking-large-language-models","title":"MathAttack: Attacking Large Language Models Towards Math Solving Ability","date":"2023-09-04","arxiv_id":"2309.01686","repositories_listed":0,"syntology":null},{"url":null,"slug":"no-train-still-gain-unleash-mathematical","title":"No Train Still Gain. Unleash Mathematical Reasoning of Large Language Models with Monte Carlo Tree Search Guided by Energy Function","date":"2023-09-01","arxiv_id":"2309.03224","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversigate-a-comprehensive-framework-for","title":"DiversiGATE: A Comprehensive Framework for Reliable Large Language Models","date":"2023-06-22","arxiv_id":"2306.13230","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-math-word-problem-solution","title":"Interpretable Math Word Problem Solution Generation Via Step-by-step Planning","date":"2023-06-01","arxiv_id":"2306.00784","repositories_listed":0,"syntology":null},{"url":null,"slug":"rcot-detecting-and-rectifying-factual","title":"RCOT: Detecting and Rectifying Factual Inconsistency in Reasoning by Reversing Chain-of-Thought","date":"2023-05-19","arxiv_id":"2305.11499","repositories_listed":0,"syntology":null},{"url":null,"slug":"selfzcot-a-self-prompt-zero-shot-cot-from","title":"Hint of Thought prompting: an explainable and zero-shot approach to reasoning tasks with LLMs","date":"2023-05-19","arxiv_id":"2305.11461","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-evaluation-guided-beam-search-for-1","title":"Self-Evaluation Guided Beam Search for Reasoning","date":"2023-05-01","arxiv_id":"2305.00633","repositories_listed":0,"syntology":null},{"url":null,"slug":"teaching-small-language-models-to-reason","title":"Teaching Small Language Models to Reason","date":"2022-12-16","arxiv_id":"2212.08410","repositories_listed":0,"syntology":null},{"url":null,"slug":"explicit-knowledge-transfer-for-weakly","title":"Explicit Knowledge Transfer for Weakly-Supervised Code Generation","date":"2022-11-30","arxiv_id":"2211.16740","repositories_listed":0,"syntology":null},{"url":"/paper/solving-math-word-problems-with-process-and","slug":"solving-math-word-problems-with-process-and","title":"Solving math word problems with process- and outcome-based feedback","date":"2022-11-25","arxiv_id":"2211.14275","repositories_listed":0,"syntology":null},{"url":"/paper/large-language-models-can-self-improve","slug":"large-language-models-can-self-improve","title":"Large Language Models Can Self-Improve","date":"2022-10-20","arxiv_id":"2210.11610","repositories_listed":0,"syntology":null},{"url":"/paper/transcending-scaling-laws-with-0-1-extra","slug":"transcending-scaling-laws-with-0-1-extra","title":"Transcending Scaling Laws with 0.1% Extra Compute","date":"2022-10-20","arxiv_id":"2210.11399","repositories_listed":0,"syntology":null},{"url":null,"slug":"complexity-based-prompting-for-multi-step","title":"Complexity-Based Prompting for Multi-Step Reasoning","date":"2022-10-03","arxiv_id":"2210.00720","repositories_listed":0,"syntology":null},{"url":"/paper/on-the-advance-of-making-language-models","slug":"on-the-advance-of-making-language-models","title":"Making Large Language Models Better Reasoners with Step-Aware Verifier","date":"2022-06-06","arxiv_id":"2206.02336","repositories_listed":0,"syntology":null}],"record_sha256":"e887995d515e0cd9771154620309f9ec898df79bfb25a5cb142be9c9169f24d3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}