{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mathematical-reasoning/papers/6","list_of":"/task/mathematical-reasoning","task":"Mathematical Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":9,"rows_per_page":100,"rows":[501,600],"of":805,"counts":{"archive_papers_tagged":805,"with_a_code_link":395,"where_syntology_ran_a_sample":197,"not_listed_spam_title":0,"listed":805,"listed_where_code_ran":197,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":159,"every_run_a_failure_of_syntologys_instrument":38,"listed_with_a_run_with_no_instrument_failure":159,"listed_every_run_a_failure_of_syntologys_instrument":38,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mathematical-reasoning","prev":"/task/mathematical-reasoning/papers/5","next":"/task/mathematical-reasoning/papers/7","papers":[{"url":null,"slug":"enhancing-mathematical-reasoning-in-large","title":"Enhancing Mathematical Reasoning in Large Language Models with Self-Consistency-Based Hallucination Detection","date":"2025-04-13","arxiv_id":"2504.09440","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-optimism-correction-be-confident","title":"Supervised Optimism Correction: Be Confident When LLMs Are Sure","date":"2025-04-10","arxiv_id":"2504.07527","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-data-generation-multi-step-rl-for","title":"Synthetic Data Generation & Multi-Step RL for Reasoning & Tool Use","date":"2025-04-07","arxiv_id":"2504.04736","repositories_listed":0,"syntology":null},{"url":null,"slug":"explain-with-visual-keypoints-like-a-real","title":"Explain with Visual Keypoints Like a Real Mentor! A Benchmark for Multimodal Solution Explanation","date":"2025-04-04","arxiv_id":"2504.03197","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-don-t-search-rethinking-test-time","title":"Sample, Don't Search: Rethinking Test-Time Alignment for Language Models","date":"2025-04-04","arxiv_id":"2504.03790","repositories_listed":0,"syntology":null},{"url":null,"slug":"lexpam-legal-procedure-awareness-guided","title":"LexPam: Legal Procedure Awareness-Guided Mathematical Reasoning","date":"2025-04-03","arxiv_id":"2504.02590","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-for-complex-reasoning-task-an-exploratory","title":"LLM for Complex Reasoning Task: An Exploratory Study in Fermi Problems","date":"2025-04-03","arxiv_id":"2504.02671","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-library-learning-fails-a-lego-prover-case","title":"LLM Library Learning Fails: A LEGO-Prover Case Study","date":"2025-04-03","arxiv_id":"2504.03048","repositories_listed":0,"syntology":null},{"url":null,"slug":"brains-vs-bytes-evaluating-llm-proficiency-in","title":"Brains vs. Bytes: Evaluating LLM Proficiency in Olympiad Mathematics","date":"2025-04-01","arxiv_id":"2504.01995","repositories_listed":0,"syntology":null},{"url":null,"slug":"genprm-scaling-test-time-compute-of-process","title":"GenPRM: Scaling Test-Time Compute of Process Reward Models via Generative Reasoning","date":"2025-04-01","arxiv_id":"2504.00891","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-difficulty-aware-staged-reinforcement","title":"How Difficulty-Aware Staged Reinforcement Learning Enhances LLMs' Reasoning Capabilities: A Preliminary Experimental Study","date":"2025-04-01","arxiv_id":"2504.00829","repositories_listed":0,"syntology":null},{"url":null,"slug":"verifiagent-a-unified-verification-agent-in","title":"VerifiAgent: a Unified Verification Agent in Language Model Reasoning","date":"2025-04-01","arxiv_id":"2504.00406","repositories_listed":0,"syntology":null},{"url":null,"slug":"expanding-rl-with-verifiable-rewards-across","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","date":"2025-03-31","arxiv_id":"2503.23829","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-axiom-based-atlas-a-structural-mapping-of","title":"The Axiom-Based Atlas: A Structural Mapping of Theorems via Foundational Proof Vectors","date":"2025-03-31","arxiv_id":"2504.00063","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-aware-branching-for-improved","title":"Entropy-Aware Branching for Improved Mathematical Reasoning","date":"2025-03-27","arxiv_id":"2503.21961","repositories_listed":0,"syntology":null},{"url":null,"slug":"proof-or-bluff-evaluating-llms-on-2025-usa","title":"Proof or Bluff? Evaluating LLMs on 2025 USA Math Olympiad","date":"2025-03-27","arxiv_id":"2503.21934","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathglance-multimodal-large-language-models","title":"MATHGLANCE: Multimodal Large Language Models Do Not Know Where to Look in Mathematical Diagrams","date":"2025-03-26","arxiv_id":"2503.20745","repositories_listed":0,"syntology":null},{"url":null,"slug":"innate-reasoning-is-not-enough-in-context","title":"Innate Reasoning is Not Enough: In-Context Learning Enhances Reasoning Large Language Models with Less Overthinking","date":"2025-03-25","arxiv_id":"2503.19602","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-chain-of-thought-with-jensen-s","title":"Learning to chain-of-thought with Jensen's evidence lower bound","date":"2025-03-25","arxiv_id":"2503.19618","repositories_listed":0,"syntology":null},{"url":null,"slug":"process-or-result-manipulated-ending-tokens","title":"Process or Result? Manipulated Ending Tokens Can Mislead Reasoning LLMs to Ignore the Correct Reasoning Steps","date":"2025-03-25","arxiv_id":"2503.19326","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-finetuning-llms-from-on-and-off-policy","title":"RL-finetuning LLMs from on- and off-policy data with a single algorithm","date":"2025-03-25","arxiv_id":"2503.19612","repositories_listed":0,"syntology":null},{"url":null,"slug":"clear-contrasting-textual-feedback-with","title":"CLEAR: Contrasting Textual Feedback with Experts and Amateurs for Reasoning","date":"2025-03-24","arxiv_id":"2504.07116","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-visual-forgetting-via-take-along","title":"Mitigating Visual Forgetting via Take-along Visual Conditioning for Multi-modal Long CoT Reasoning","date":"2025-03-17","arxiv_id":"2503.13360","repositories_listed":0,"syntology":null},{"url":null,"slug":"pensez-less-data-better-reasoning-rethinking","title":"Pensez: Less Data, Better Reasoning -- Rethinking French LLM","date":"2025-03-17","arxiv_id":"2503.13661","repositories_listed":0,"syntology":null},{"url":null,"slug":"reliable-and-efficient-amortized-model-based","title":"Reliable and Efficient Amortized Model-based Evaluation","date":"2025-03-17","arxiv_id":"2503.13335","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-the-mathematical-reasoning-in","title":"Evaluating Mathematical Reasoning Across Large Language Models: A Fine-Grained Approach","date":"2025-03-13","arxiv_id":"2503.10573","repositories_listed":0,"syntology":null},{"url":"/paper/pi-gps-enhancing-geometry-problem-solving-by","slug":"pi-gps-enhancing-geometry-problem-solving-by","title":"Pi-GPS: Enhancing Geometry Problem Solving by Unleashing the Power of Diagrammatic Information","date":"2025-03-07","arxiv_id":"2503.05543","repositories_listed":0,"syntology":{"n":7,"n_ran":7,"n_constructed":1,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":3,"n_no_contract":1,"n_pointer_only":7,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 3 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pi-gps-enhancing-geometry-problem-solving-by#ran","syntology_url":"https://syntology.ai/paper/2503.05543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05543"}},"official":null}},{"url":null,"slug":"speculative-decoding-for-multi-sample","title":"Speculative Decoding for Multi-Sample Inference","date":"2025-03-07","arxiv_id":"2503.05330","repositories_listed":0,"syntology":null},{"url":null,"slug":"better-process-supervision-with-bi","title":"Better Process Supervision with Bi-directional Rewarding Signals","date":"2025-03-06","arxiv_id":"2503.04618","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-multi-round-large","title":"Towards Understanding Multi-Round Large Language Model Reasoning: Approximability, Learnability and Generalizability","date":"2025-03-05","arxiv_id":"2503.03128","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-evolved-preference-optimization-for","title":"Self-Evolved Preference Optimization for Enhancing Mathematical Reasoning in Small Language Models","date":"2025-03-04","arxiv_id":"2503.04813","repositories_listed":0,"syntology":null},{"url":null,"slug":"none-of-the-above-less-of-the-right-parallel","title":"None of the Above, Less of the Right: Parallel Patterns between Humans and LLMs on Multi-Choice Questions Answering","date":"2025-03-03","arxiv_id":"2503.01550","repositories_listed":0,"syntology":null},{"url":null,"slug":"mv-math-evaluating-multimodal-math-reasoning","title":"MV-MATH: Evaluating Multimodal Math Reasoning in Multi-Visual Contexts","date":"2025-02-28","arxiv_id":"2502.20808","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reasoner-dynamic-guidance-for-optimized","title":"Meta-Reasoner: Dynamic Guidance for Optimized Inference-time Reasoning in Large Language Models","date":"2025-02-27","arxiv_id":"2502.19918","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-2-multi-agent-test-time-scalable","title":"Multi2: Multi-Agent Test-Time Scalable Framework for Multi-Document Processing","date":"2025-02-27","arxiv_id":"2502.20592","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-self-consistency-from-dynamic","title":"Revisiting Self-Consistency from Dynamic Distributional Alignment Perspective on Answer Aggregation","date":"2025-02-27","arxiv_id":"2502.19830","repositories_listed":0,"syntology":null},{"url":null,"slug":"thinking-slow-fast-scaling-inference-compute","title":"Thinking Slow, Fast: Scaling Inference Compute with Distilled Reasoners","date":"2025-02-27","arxiv_id":"2502.20339","repositories_listed":0,"syntology":null},{"url":null,"slug":"weaker-llms-opinions-also-matter-mixture-of","title":"Weaker LLMs' Opinions Also Matter: Mixture of Opinions Enhances LLM's Mathematical Reasoning","date":"2025-02-26","arxiv_id":"2502.19622","repositories_listed":0,"syntology":null},{"url":null,"slug":"leanprogress-guiding-search-for-neural","title":"LeanProgress: Guiding Search for Neural Theorem Proving via Proof Progress Prediction","date":"2025-02-25","arxiv_id":"2502.17925","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-thinking-optimal-scaling-of-test-time","title":"Towards Thinking-Optimal Scaling of Test-Time Compute for LLM Reasoning","date":"2025-02-25","arxiv_id":"2502.18080","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-step-dpo-self-supervised-preference","title":"Full-Step-DPO: Self-Supervised Preference Optimization with Step-wise Rewards for Mathematical Reasoning","date":"2025-02-20","arxiv_id":"2502.14356","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-augmented-process-reward-model-for","title":"Retrieval-Augmented Process Reward Model for Generalizable Mathematical Reasoning","date":"2025-02-20","arxiv_id":"2502.14361","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-correctness-to-comprehension-ai-agents","title":"From Correctness to Comprehension: AI Agents for Personalized Error Diagnosis in Education","date":"2025-02-19","arxiv_id":"2502.13789","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-arithmetic-learning-improves","title":"Integrating Arithmetic Learning Improves Mathematical Reasoning in Smaller Models","date":"2025-02-18","arxiv_id":"2502.12855","repositories_listed":0,"syntology":null},{"url":null,"slug":"sens-merging-sensitivity-guided-parameter","title":"Sens-Merging: Sensitivity-Guided Parameter Balancing for Merging Large Language Models","date":"2025-02-18","arxiv_id":"2502.12420","repositories_listed":0,"syntology":null},{"url":null,"slug":"theorem-prover-as-a-judge-for-synthetic-data","title":"Theorem Prover as a Judge for Synthetic Data Generation","date":"2025-02-18","arxiv_id":"2502.13137","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-and-mathematical","title":"Large Language Models and Mathematical Reasoning Failures","date":"2025-02-17","arxiv_id":"2502.11574","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathfimer-enhancing-mathematical-reasoning-by","title":"MathFimer: Enhancing Mathematical Reasoning by Expanding Reasoning Steps through Fill-in-the-Middle Task","date":"2025-02-17","arxiv_id":"2502.11684","repositories_listed":0,"syntology":null},{"url":null,"slug":"teaching-llms-according-to-their-aptitude","title":"Teaching LLMs According to Their Aptitude: Adaptive Reasoning for Mathematical Problem Solving","date":"2025-02-17","arxiv_id":"2502.12022","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-constrained-monte-carlo-tree","title":"Leveraging Constrained Monte Carlo Tree Search to Generate Reliable Long Chain-of-Thought for Mathematical Reasoning","date":"2025-02-16","arxiv_id":"2502.11169","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-step-wise-verification-with","title":"Uncertainty-Aware Step-wise Verification with Generative Reward Models","date":"2025-02-16","arxiv_id":"2502.11250","repositories_listed":0,"syntology":null},{"url":null,"slug":"1bit-merging-dynamic-quantized-merging-for","title":"1bit-Merging: Dynamic Quantized Merging for Large Language Models","date":"2025-02-15","arxiv_id":"2502.10743","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-meta-and-object-level","title":"Evaluating the Meta- and Object-Level Reasoning of Large Language Models for Question Answering","date":"2025-02-14","arxiv_id":"2502.10338","repositories_listed":0,"syntology":null},{"url":null,"slug":"gora-gradient-driven-adaptive-low-rank","title":"GoRA: Gradient-driven Adaptive Low Rank Adaptation","date":"2025-02-13","arxiv_id":"2502.12171","repositories_listed":0,"syntology":null},{"url":null,"slug":"llms-can-implicitly-learn-from-mistakes-in","title":"LLMs can implicitly learn from mistakes in-context","date":"2025-02-12","arxiv_id":"2502.08550","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-example-shown-many-concepts-known","title":"One Example Shown, Many Concepts Known! Counterexample-Driven Conceptual Reasoning in Mathematical LLMs","date":"2025-02-12","arxiv_id":"2502.10454","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-self-to-supervised-fine-tuning-for","title":"Selective Self-to-Supervised Fine-Tuning for Generalization in Large Language Models","date":"2025-02-12","arxiv_id":"2502.08130","repositories_listed":0,"syntology":null},{"url":null,"slug":"math-perturb-benchmarking-llms-math-reasoning","title":"MATH-Perturb: Benchmarking LLMs' Math Reasoning Abilities against Hard Perturbations","date":"2025-02-10","arxiv_id":"2502.06453","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-training-large-language-models-for-tool","title":"Self-Training Large Language Models for Tool-Use Without Demonstrations","date":"2025-02-09","arxiv_id":"2502.05867","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolving-llms-self-refinement-capability-via","title":"Evolving LLMs' Self-Refinement Capability via Iterative Preference Optimization","date":"2025-02-08","arxiv_id":"2502.05605","repositories_listed":0,"syntology":null},{"url":null,"slug":"llms-can-be-easily-confused-by-instructional","title":"LLMs can be easily Confused by Instructional Distractions","date":"2025-02-05","arxiv_id":"2502.04362","repositories_listed":0,"syntology":null},{"url":null,"slug":"path-planning-for-masked-diffusion-model","title":"Path Planning for Masked Diffusion Model Sampling","date":"2025-02-05","arxiv_id":"2502.03540","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-as-logic-units-scaling-test-time","title":"Reasoning-as-Logic-Units: Scaling Test-Time Reasoning in Large Language Models Through Logic Unit Alignment","date":"2025-02-05","arxiv_id":"2502.07803","repositories_listed":0,"syntology":null},{"url":null,"slug":"token-assorted-mixing-latent-and-text-tokens","title":"Token Assorted: Mixing Latent and Text Tokens for Improved Language Model Reasoning","date":"2025-02-05","arxiv_id":"2502.03275","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-guided-tree-search-for-enhanced-llm","title":"Policy Guided Tree Search for Enhanced LLM Reasoning","date":"2025-02-04","arxiv_id":"2502.06813","repositories_listed":0,"syntology":null},{"url":null,"slug":"premise-augmented-reasoning-chains-improve","title":"Premise-Augmented Reasoning Chains Improve Error Identification in Math reasoning with LLMs","date":"2025-02-04","arxiv_id":"2502.02362","repositories_listed":0,"syntology":null},{"url":null,"slug":"satori-reinforcement-learning-with-chain-of","title":"Satori: Reinforcement Learning with Chain-of-Action-Thought Enhances LLM Reasoning via Autoregressive Search","date":"2025-02-04","arxiv_id":"2502.02508","repositories_listed":0,"syntology":null},{"url":null,"slug":"mergeme-model-merging-techniques-for","title":"MergeME: Model Merging Techniques for Homogeneous and Heterogeneous MoEs","date":"2025-02-03","arxiv_id":"2502.00997","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-models-use-trigonometry-to-do","title":"Language Models Use Trigonometry to Do Addition","date":"2025-02-02","arxiv_id":"2502.00873","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rule-based-reasoning-in-llms-via","title":"Improving Rule-based Reasoning in LLMs via Neurosymbolic Representations","date":"2025-01-31","arxiv_id":"2502.01657","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-informal-to-formal-incorporating-and","title":"From Informal to Formal -- Incorporating and Evaluating LLMs on Natural Language Requirements to Verifiable Formal Proofs","date":"2025-01-27","arxiv_id":"2501.16207","repositories_listed":0,"syntology":null},{"url":null,"slug":"lemmahead-rag-assisted-proof-generation-using","title":"LemmaHead: RAG Assisted Proof Generation Using Large Language Models","date":"2025-01-27","arxiv_id":"2501.15797","repositories_listed":0,"syntology":null},{"url":null,"slug":"error-classification-of-large-language-models","title":"Error Classification of Large Language Models on Math Word Problems: A Dynamically Adaptive Framework","date":"2025-01-26","arxiv_id":"2501.15581","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-karp-dataset","title":"The Karp Dataset","date":"2025-01-24","arxiv_id":"2501.14705","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-math-reasoning-in-language-models","title":"Advancing Mathematical Reasoning in Language Models: The Impact of Problem-Solving Data, Data Synthesis Methods, and Training Stages","date":"2025-01-23","arxiv_id":"2501.14002","repositories_listed":0,"syntology":null},{"url":null,"slug":"coarse-to-fine-process-reward-modeling-for","title":"Coarse-to-Fine Process Reward Modeling for Enhanced Mathematical Reasoning","date":"2025-01-23","arxiv_id":"2501.13622","repositories_listed":0,"syntology":null},{"url":"/paper/ugmathbench-a-diverse-and-dynamic-benchmark","slug":"ugmathbench-a-diverse-and-dynamic-benchmark","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","date":"2025-01-23","arxiv_id":"2501.13766","repositories_listed":0,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/ugmathbench-a-diverse-and-dynamic-benchmark#ran","syntology_url":"https://syntology.ai/paper/2501.13766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.13766"}},"official":null}},{"url":null,"slug":"cdw-cot-clustered-distance-weighted-chain-of","title":"CDW-CoT: Clustered Distance-Weighted Chain-of-Thoughts Reasoning","date":"2025-01-21","arxiv_id":"2501.12226","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-language-models-via-random","title":"Benchmarking Large Language Models via Random Variables","date":"2025-01-20","arxiv_id":"2501.11790","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-reasoning-towards-unified","title":"Chain-of-Reasoning: Towards Unified Mathematical Reasoning in Large Language Models via a Multi-Paradigm Perspective","date":"2025-01-19","arxiv_id":"2501.11110","repositories_listed":0,"syntology":null},{"url":null,"slug":"step-kto-optimizing-mathematical-reasoning","title":"Step-KTO: Optimizing Mathematical Reasoning through Stepwise Binary Feedback","date":"2025-01-18","arxiv_id":"2501.10799","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-lessons-of-developing-process-reward","title":"The Lessons of Developing Process Reward Models in Mathematical Reasoning","date":"2025-01-13","arxiv_id":"2501.07301","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantization-meets-reasoning-exploring-llm","title":"Quantization Meets Reasoning: Exploring LLM Low-Bit Quantization Degradation for Mathematical Reasoning","date":"2025-01-06","arxiv_id":"2501.03035","repositories_listed":0,"syntology":null},{"url":null,"slug":"understand-solve-and-translate-bridging-the","title":"Understand, Solve and Translate: Bridging the Multilingual Mathematical Reasoning Gap","date":"2025-01-05","arxiv_id":"2501.02448","repositories_listed":0,"syntology":null},{"url":null,"slug":"table-as-thought-exploring-structured","title":"Table as Thought: Exploring Structured Thoughts in LLM Reasoning","date":"2025-01-04","arxiv_id":"2501.02152","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-reasoning-through-process","title":"Enhancing Reasoning through Process Supervision with Monte Carlo Tree Search","date":"2025-01-02","arxiv_id":"2501.01478","repositories_listed":0,"syntology":null},{"url":null,"slug":"plug-and-play-training-framework-for","title":"Plug-and-Play Training Framework for Preference Optimization","date":"2024-12-30","arxiv_id":"2412.20996","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-reasoning-engine-specialized-training-for","title":"LLM Reasoning Engine: Specialized Training for Enhanced Mathematical Reasoning","date":"2024-12-28","arxiv_id":"2412.20227","repositories_listed":0,"syntology":null},{"url":null,"slug":"system-2-mathematical-reasoning-via-enriched","title":"System-2 Mathematical Reasoning via Enriched Instruction Tuning","date":"2024-12-22","arxiv_id":"2412.16964","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensembling-large-language-models-with-process","title":"Ensembling Large Language Models with Process Reward-Guided Tree Search for Better Complex Reasoning","date":"2024-12-20","arxiv_id":"2412.15797","repositories_listed":0,"syntology":null},{"url":null,"slug":"formal-mathematical-reasoning-a-new-frontier","title":"Formal Mathematical Reasoning: A New Frontier in AI","date":"2024-12-20","arxiv_id":"2412.16075","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-are-step-level-reward-models-rewarding","title":"What Are Step-Level Reward Models Rewarding? Counterintuitive Findings from MCTS-Boosted Mathematical Reasoning","date":"2024-12-20","arxiv_id":"2412.15904","repositories_listed":0,"syntology":null},{"url":null,"slug":"channel-merging-preserving-specialization-for","title":"Channel Merging: Preserving Specialization for Merged Experts","date":"2024-12-18","arxiv_id":"2412.15283","repositories_listed":0,"syntology":null},{"url":null,"slug":"metarulegpt-recursive-numerical-reasoning-of","title":"MetaRuleGPT: Recursive Numerical Reasoning of Language Models Trained with Simple Rules","date":"2024-12-18","arxiv_id":"2412.13536","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-mathematical-reasoning-in-the-era","title":"A Survey of Mathematical Reasoning in the Era of Multimodal Large Language Model: Benchmark, Method & Challenges","date":"2024-12-16","arxiv_id":"2412.11936","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-language-models-rival-mathematics","title":"Can Language Models Rival Mathematics Students? Evaluating Mathematical Reasoning through Textual Manipulation and Human Experiments","date":"2024-12-16","arxiv_id":"2412.11908","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-rank-adaptation-with-task-relevant","title":"Low-Rank Adaptation with Task-Relevant Feature Enhancement for Fine-tuning Language Models","date":"2024-12-13","arxiv_id":"2412.09827","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-graph-based-synthetic-data-pipeline-for","title":"A Graph-Based Synthetic Data Pipeline for Scaling High-Quality Reasoning Instructions","date":"2024-12-12","arxiv_id":"2412.08864","repositories_listed":0,"syntology":null},{"url":null,"slug":"sail-into-the-headwind-alignment-via-robust","title":"Sail into the Headwind: Alignment via Robust Rewards and Dynamic Labels against Reward Hacking","date":"2024-12-12","arxiv_id":"2412.09544","repositories_listed":0,"syntology":null},{"url":null,"slug":"smoltulu-higher-learning-rate-to-batch-size","title":"SmolTulu: Higher Learning Rate to Batch Size Ratios Can Lead to Better Reasoning in SLMs","date":"2024-12-11","arxiv_id":"2412.08347","repositories_listed":0,"syntology":null}],"record_sha256":"7816437790a75f5213aeb66dac9a62432c5293d00a85f0b5642eefc1e5c3f76e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}