{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/math/papers/10","list_of":"/task/math","task":"Math","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":16,"rows_per_page":100,"rows":[901,1000],"of":1596,"counts":{"archive_papers_tagged":1596,"with_a_code_link":765,"where_syntology_ran_a_sample":349,"not_listed_spam_title":0,"listed":1596,"listed_where_code_ran":349,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/math","prev":"/task/math/papers/9","next":"/task/math/papers/11","papers":[{"url":null,"slug":"reasoning-models-know-when-they-re-right","title":"Reasoning Models Know When They're Right: Probing Hidden States for Self-Verification","date":"2025-04-07","arxiv_id":"2504.05419","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-data-generation-multi-step-rl-for","title":"Synthetic Data Generation & Multi-Step RL for Reasoning & Tool Use","date":"2025-04-07","arxiv_id":"2504.04736","repositories_listed":0,"syntology":null},{"url":null,"slug":"retro-search-exploring-untaken-paths-for","title":"Retro-Search: Exploring Untaken Paths for Deeper and Efficient Reasoning","date":"2025-04-06","arxiv_id":"2504.04383","repositories_listed":0,"syntology":null},{"url":null,"slug":"onedal-optimization-for-arm-scalable-vector","title":"oneDAL Optimization for ARM Scalable Vector Extension: Maximizing Efficiency for High-Performance Data Science","date":"2025-04-05","arxiv_id":"2504.04241","repositories_listed":0,"syntology":null},{"url":null,"slug":"explain-with-visual-keypoints-like-a-real","title":"Explain with Visual Keypoints Like a Real Mentor! A Benchmark for Multimodal Solution Explanation","date":"2025-04-04","arxiv_id":"2504.03197","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-difficulty-filtering-for-reasoning","title":"Online Difficulty Filtering for Reasoning Oriented Reinforcement Learning","date":"2025-04-04","arxiv_id":"2504.03380","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-consistency-a-novel-inference","title":"Cross-Lingual Consistency: A Novel Inference Framework for Advancing Reasoning in Large Language Models","date":"2025-04-02","arxiv_id":"2504.01857","repositories_listed":0,"syntology":null},{"url":null,"slug":"brains-vs-bytes-evaluating-llm-proficiency-in","title":"Brains vs. Bytes: Evaluating LLM Proficiency in Olympiad Mathematics","date":"2025-04-01","arxiv_id":"2504.01995","repositories_listed":0,"syntology":null},{"url":null,"slug":"genprm-scaling-test-time-compute-of-process","title":"GenPRM: Scaling Test-Time Compute of Process Reward Models via Generative Reasoning","date":"2025-04-01","arxiv_id":"2504.00891","repositories_listed":0,"syntology":null},{"url":null,"slug":"hawkeye-efficient-reasoning-with-model","title":"Hawkeye:Efficient Reasoning with Model Collaboration","date":"2025-04-01","arxiv_id":"2504.00424","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-difficulty-aware-staged-reinforcement","title":"How Difficulty-Aware Staged Reinforcement Learning Enhances LLMs' Reasoning Capabilities: A Preliminary Experimental Study","date":"2025-04-01","arxiv_id":"2504.00829","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-large-language-models-in-1","title":"Investigating Large Language Models in Diagnosing Students' Cognitive Skills in Math Problem-solving","date":"2025-04-01","arxiv_id":"2504.00843","repositories_listed":0,"syntology":null},{"url":null,"slug":"debflow-automating-agent-creation-via-agent","title":"DebFlow: Automating Agent Creation via Agent Debate","date":"2025-03-31","arxiv_id":"2503.23781","repositories_listed":0,"syntology":null},{"url":null,"slug":"proof-or-bluff-evaluating-llms-on-2025-usa","title":"Proof or Bluff? Evaluating LLMs on 2025 USA Math Olympiad","date":"2025-03-27","arxiv_id":"2503.21934","repositories_listed":0,"syntology":null},{"url":null,"slug":"1-4-million-open-source-distilled-reasoning","title":"1.4 Million Open-Source Distilled Reasoning Dataset to Empower Large Language Model Training","date":"2025-03-25","arxiv_id":"2503.19633","repositories_listed":0,"syntology":null},{"url":null,"slug":"gemma-3-technical-report","title":"Gemma 3 Technical Report","date":"2025-03-25","arxiv_id":"2503.19786","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-evaluation-time-compute-with","title":"Scaling Evaluation-time Compute with Reasoning Models as Process Evaluators","date":"2025-03-25","arxiv_id":"2503.19877","repositories_listed":0,"syntology":null},{"url":null,"slug":"think-twice-enhancing-llm-reasoning-by","title":"Think Twice: Enhancing LLM Reasoning by Scaling Multi-round Test-time Thinking","date":"2025-03-25","arxiv_id":"2503.19855","repositories_listed":0,"syntology":null},{"url":null,"slug":"activation-functions-considered-harmful","title":"Activation Functions Considered Harmful: Recovering Neural Network Weights through Controlled Channels","date":"2025-03-24","arxiv_id":"2503.19142","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-vocabulary-mismatch-vocabulary","title":"Overcoming Vocabulary Mismatch: Vocabulary-agnostic Teacher Guided Language Modeling","date":"2025-03-24","arxiv_id":"2503.19123","repositories_listed":0,"syntology":null},{"url":null,"slug":"teaching-llms-for-step-level-automatic-math","title":"Teaching LLMs for Step-Level Automatic Math Correction via Reinforcement Learning","date":"2025-03-24","arxiv_id":"2503.18432","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-is-more-important-than-difficult-for","title":"Long Is More Important Than Difficult for Training Reasoning Models","date":"2025-03-23","arxiv_id":"2503.18069","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathagent-leveraging-a-mixture-of-math-agent","title":"MathAgent: Leveraging a Mixture-of-Math-Agent Framework for Real-World Multimodal Mathematical Error Detection","date":"2025-03-23","arxiv_id":"2503.18132","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-hidden-reasoning-process-of","title":"Exploring the Hidden Reasoning Process of Large Language Models by Misleading Them","date":"2025-03-20","arxiv_id":"2503.16401","repositories_listed":0,"syntology":null},{"url":null,"slug":"tapered-off-policy-reinforce-stable-and","title":"Tapered Off-Policy REINFORCE: Stable and efficient reinforcement learning for LLMs","date":"2025-03-18","arxiv_id":"2503.14286","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-complex-reasoning-with-dynamic","title":"Improving Complex Reasoning with Dynamic Prompt Corruption: A soft prompt Optimization Approach","date":"2025-03-17","arxiv_id":"2503.13208","repositories_listed":0,"syntology":null},{"url":null,"slug":"pensez-less-data-better-reasoning-rethinking","title":"Pensez: Less Data, Better Reasoning -- Rethinking French LLM","date":"2025-03-17","arxiv_id":"2503.13661","repositories_listed":0,"syntology":null},{"url":null,"slug":"spin-bench-how-well-do-llms-plan","title":"SPIN-Bench: How Well Do LLMs Plan Strategically and Reason Socially?","date":"2025-03-16","arxiv_id":"2503.12349","repositories_listed":0,"syntology":null},{"url":null,"slug":"chat-ts-enhancing-multi-modal-reasoning-over","title":"Chat-TS: Enhancing Multi-Modal Reasoning Over Time-Series and Natural Language Data","date":"2025-03-13","arxiv_id":"2503.10883","repositories_listed":0,"syntology":null},{"url":null,"slug":"conformal-prediction-sets-for-deep-generative","title":"Conformal Prediction Sets for Deep Generative Models via Reduction to Conformal Regression","date":"2025-03-13","arxiv_id":"2503.10512","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-item-writing-flaws-on","title":"The Impact of Item-Writing Flaws on Difficulty and Discrimination in Item Response Theory","date":"2025-03-13","arxiv_id":"2503.10533","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-logical-capabilities-of","title":"Understanding the Logical Capabilities of Large Language Models via Out-of-Context Representation Learning","date":"2025-03-13","arxiv_id":"2503.10408","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-text-to-visuals-using-llms-to-generate","title":"From Text to Visuals: Using LLMs to Generate Math Diagrams with Vector Graphics","date":"2025-03-10","arxiv_id":"2503.07429","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-test-time-compute-via-meta","title":"Optimizing Test-Time Compute via Meta Reinforcement Fine-Tuning","date":"2025-03-10","arxiv_id":"2503.07572","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-the-black-box-integrating-moral","title":"Decoding the Black Box: Integrating Moral Imagination with Technical AI Governance","date":"2025-03-09","arxiv_id":"2503.06411","repositories_listed":0,"syntology":null},{"url":null,"slug":"inftythink-breaking-the-length-limits-of-long","title":"InftyThink: Breaking the Length Limits of Long-Context Reasoning in Large Language Models","date":"2025-03-09","arxiv_id":"2503.06692","repositories_listed":0,"syntology":null},{"url":null,"slug":"symbolic-mixture-of-experts-adaptive-skill","title":"Symbolic Mixture-of-Experts: Adaptive Skill-based Routing for Heterogeneous Reasoning","date":"2025-03-07","arxiv_id":"2503.05641","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-reasoning-robustness-in-large","title":"Benchmarking Reasoning Robustness in Large Language Models","date":"2025-03-06","arxiv_id":"2503.04550","repositories_listed":0,"syntology":null},{"url":null,"slug":"better-process-supervision-with-bi","title":"Better Process Supervision with Bi-directional Rewarding Signals","date":"2025-03-06","arxiv_id":"2503.04618","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-causal-reasoning-evaluation-in","title":"Compositional Causal Reasoning Evaluation in Language Models","date":"2025-03-06","arxiv_id":"2503.04556","repositories_listed":0,"syntology":null},{"url":null,"slug":"helpsteer3-human-annotated-feedback-and-edit","title":"HelpSteer3: Human-Annotated Feedback and Edit Data to Empower Inference-Time Scaling in Open-Ended General-Domain Tasks","date":"2025-03-06","arxiv_id":"2503.04378","repositories_listed":0,"syntology":null},{"url":null,"slug":"solar-scalable-optimization-of-large-scale","title":"SOLAR: Scalable Optimization of Large-scale Architecture for Reasoning","date":"2025-03-06","arxiv_id":"2503.04530","repositories_listed":0,"syntology":null},{"url":null,"slug":"start-self-taught-reasoner-with-tools","title":"START: Self-taught Reasoner with Tools","date":"2025-03-06","arxiv_id":"2503.04625","repositories_listed":0,"syntology":null},{"url":null,"slug":"fans-formal-answer-selection-for-natural","title":"FANS -- Formal Answer Selection for Natural Language Math Reasoning Using Lean4","date":"2025-03-05","arxiv_id":"2503.03238","repositories_listed":0,"syntology":null},{"url":null,"slug":"lewis-layer-wise-sparsity-a-training-free","title":"LEWIS (LayEr WIse Sparsity) -- A Training Free Guided Model Merging Approach","date":"2025-03-05","arxiv_id":"2503.03874","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-comparison-of-large-language-1","title":"Performance Comparison of Large Language Models on Advanced Calculus Problems","date":"2025-03-05","arxiv_id":"2503.03960","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-evolved-preference-optimization-for","title":"Self-Evolved Preference Optimization for Enhancing Mathematical Reasoning in Small Language Models","date":"2025-03-04","arxiv_id":"2503.04813","repositories_listed":0,"syntology":null},{"url":null,"slug":"cats-confuse-reasoning-llm-query-agnostic","title":"Cats Confuse Reasoning LLM: Query Agnostic Adversarial Triggers for Reasoning Models","date":"2025-03-03","arxiv_id":"2503.01781","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-s-behind-ppo-s-collapse-in-long-cot","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","date":"2025-03-03","arxiv_id":"2503.01491","repositories_listed":0,"syntology":null},{"url":null,"slug":"mv-math-evaluating-multimodal-math-reasoning","title":"MV-MATH: Evaluating Multimodal Math Reasoning in Multi-Visual Contexts","date":"2025-02-28","arxiv_id":"2502.20808","repositories_listed":0,"syntology":null},{"url":null,"slug":"med-rlvr-emerging-medical-reasoning-from-a-3b","title":"Med-RLVR: Emerging Medical Reasoning from a 3B base model via reinforcement Learning","date":"2025-02-27","arxiv_id":"2502.19655","repositories_listed":0,"syntology":null},{"url":null,"slug":"swe-rl-advancing-llm-reasoning-via","title":"SWE-RL: Advancing LLM Reasoning via Reinforcement Learning on Open Software Evolution","date":"2025-02-25","arxiv_id":"2502.18449","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-thinking-optimal-scaling-of-test-time","title":"Towards Thinking-Optimal Scaling of Test-Time Compute for LLM Reasoning","date":"2025-02-25","arxiv_id":"2502.18080","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-decentralized-swarms-using-rotation","title":"Learning Decentralized Swarms Using Rotation Equivariant Graph Neural Networks","date":"2025-02-24","arxiv_id":"2502.17612","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-with-latent-thoughts-on-the-power","title":"Reasoning with Latent Thoughts: On the Power of Looped Transformers","date":"2025-02-24","arxiv_id":"2502.17416","repositories_listed":0,"syntology":null},{"url":null,"slug":"disc-dynamic-decomposition-improves-llm","title":"DISC: DISC: Dynamic Decomposition Improves LLM Inference Scaling","date":"2025-02-23","arxiv_id":"2502.16706","repositories_listed":0,"syntology":null},{"url":null,"slug":"sbsc-step-by-step-coding-for-improving","title":"SBSC: Step-By-Step Coding for Improving Mathematical Olympiad Performance","date":"2025-02-23","arxiv_id":"2502.16666","repositories_listed":0,"syntology":null},{"url":null,"slug":"inference-computation-scaling-for-feature","title":"Inference Computation Scaling for Feature Augmentation in Recommendation Systems","date":"2025-02-22","arxiv_id":"2502.16040","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-reasoning-introduce-bias-a-study-of","title":"Does Reasoning Introduce Bias? A Study of Social Bias Evaluation and Mitigation in LLM Reasoning","date":"2025-02-21","arxiv_id":"2502.15361","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-feedback-based-multi-step","title":"A Survey on Feedback-based Multi-step Reasoning for Large Language Models on Mathematics","date":"2025-02-20","arxiv_id":"2502.14333","repositories_listed":0,"syntology":null},{"url":null,"slug":"beamlora-beam-constraint-low-rank-adaptation","title":"BeamLoRA: Beam-Constraint Low-Rank Adaptation","date":"2025-02-19","arxiv_id":"2502.13604","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffsampling-enhancing-diversity-and-accuracy","title":"DiffSampling: Enhancing Diversity and Accuracy in Neural Text Generation","date":"2025-02-19","arxiv_id":"2502.14037","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-self-improvement-paradox-can-language","title":"The Self-Improvement Paradox: Can Language Models Bootstrap Reasoning Capabilities without External Scaffolding?","date":"2025-02-19","arxiv_id":"2502.13441","repositories_listed":0,"syntology":null},{"url":null,"slug":"lean-ing-on-quality-how-high-quality-data","title":"Lean-ing on Quality: How High-Quality Data Beats Diverse Multilingual Data in AutoFormalization","date":"2025-02-18","arxiv_id":"2502.15795","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-step-alignment-as-markov-games-an","title":"Multi-Step Alignment as Markov Games: An Optimistic Online Gradient Descent Approach with Convergence Guarantees","date":"2025-02-18","arxiv_id":"2502.12678","repositories_listed":0,"syntology":null},{"url":null,"slug":"naturalreasoning-reasoning-in-the-wild-with-2","title":"NaturalReasoning: Reasoning in the Wild with 2.8M Challenging Questions","date":"2025-02-18","arxiv_id":"2502.13124","repositories_listed":0,"syntology":null},{"url":null,"slug":"none-of-the-others-a-general-technique-to","title":"None of the Others: a General Technique to Distinguish Reasoning from Memorization in Multiple-Choice LLM Evaluation Benchmarks","date":"2025-02-18","arxiv_id":"2502.12896","repositories_listed":0,"syntology":null},{"url":null,"slug":"thinking-outside-the-gray-box-a-context-based","title":"Thinking Outside the (Gray) Box: A Context-Based Score for Assessing Value and Originality in Neural Text Generation","date":"2025-02-18","arxiv_id":"2502.13207","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-leveraging-search-and-self","title":"A Study on Leveraging Search and Self-Feedback for Agent Reasoning","date":"2025-02-17","arxiv_id":"2502.12094","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-conscious-llm-decoding-impact-of-text","title":"Energy-Conscious LLM Decoding: Impact of Text Generation Strategies on GPU Energy Consumption","date":"2025-02-17","arxiv_id":"2502.11723","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypothesis-driven-theory-of-mind-reasoning","title":"Hypothesis-Driven Theory-of-Mind Reasoning for Large Language Models","date":"2025-02-17","arxiv_id":"2502.11881","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathfimer-enhancing-mathematical-reasoning-by","title":"MathFimer: Enhancing Mathematical Reasoning by Expanding Reasoning Steps through Fill-in-the-Middle Task","date":"2025-02-17","arxiv_id":"2502.11684","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-test-time-compute-without","title":"Scaling Test-Time Compute Without Verification or RL is Suboptimal","date":"2025-02-17","arxiv_id":"2502.12118","repositories_listed":0,"syntology":null},{"url":null,"slug":"teaching-llms-according-to-their-aptitude","title":"Teaching LLMs According to Their Aptitude: Adaptive Reasoning for Mathematical Problem Solving","date":"2025-02-17","arxiv_id":"2502.12022","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-vision-language-models-struggle-with","title":"Why Vision Language Models Struggle with Visual Arithmetic? Towards Enhanced Chart and Geometry Understanding","date":"2025-02-17","arxiv_id":"2502.11492","repositories_listed":0,"syntology":null},{"url":null,"slug":"graders-should-cheat-privileged-information","title":"Graders should cheat: privileged information enables expert-level automated evaluations","date":"2025-02-16","arxiv_id":"2502.10961","repositories_listed":0,"syntology":null},{"url":null,"slug":"1bit-merging-dynamic-quantized-merging-for","title":"1bit-Merging: Dynamic Quantized Merging for Large Language Models","date":"2025-02-15","arxiv_id":"2502.10743","repositories_listed":0,"syntology":null},{"url":null,"slug":"crane-reasoning-with-constrained-llm","title":"CRANE: Reasoning with constrained LLM generation","date":"2025-02-13","arxiv_id":"2502.09061","repositories_listed":0,"syntology":null},{"url":null,"slug":"mme-cot-benchmarking-chain-of-thought-in","title":"MME-CoT: Benchmarking Chain-of-Thought in Large Multimodal Models for Reasoning Quality, Robustness, and Efficiency","date":"2025-02-13","arxiv_id":"2502.09621","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-sketchpad-a-multimodal-tutoring","title":"Interactive Sketchpad: A Multimodal Tutoring System for Collaborative, Visual Problem-Solving","date":"2025-02-12","arxiv_id":"2503.16434","repositories_listed":0,"syntology":null},{"url":null,"slug":"o1-embedder-let-retrievers-think-before","title":"O1 Embedder: Let Retrievers Think Before Action","date":"2025-02-11","arxiv_id":"2502.07555","repositories_listed":0,"syntology":null},{"url":null,"slug":"math-perturb-benchmarking-llms-math-reasoning","title":"MATH-Perturb: Benchmarking LLMs' Math Reasoning Abilities against Hard Perturbations","date":"2025-02-10","arxiv_id":"2502.06453","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolving-llms-self-refinement-capability-via","title":"Evolving LLMs' Self-Refinement Capability via Iterative Preference Optimization","date":"2025-02-08","arxiv_id":"2502.05605","repositories_listed":0,"syntology":null},{"url":null,"slug":"bolt-bootstrap-long-chain-of-thought-in","title":"BOLT: Bootstrap Long Chain-of-Thought in Language Models without Distillation","date":"2025-02-06","arxiv_id":"2502.03860","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-adaptive-decoding-dynamic-model","title":"Entropy Adaptive Decoding: Dynamic Model Switching for Efficient Inference","date":"2025-02-05","arxiv_id":"2502.06833","repositories_listed":0,"syntology":null},{"url":null,"slug":"gold-medalist-performance-in-solving-olympiad","title":"Gold-medalist Performance in Solving Olympiad Geometry with AlphaGeometry2","date":"2025-02-05","arxiv_id":"2502.03544","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-as-logic-units-scaling-test-time","title":"Reasoning-as-Logic-Units: Scaling Test-Time Reasoning in Large Language Models Through Logic Unit Alignment","date":"2025-02-05","arxiv_id":"2502.07803","repositories_listed":0,"syntology":null},{"url":null,"slug":"premise-augmented-reasoning-chains-improve","title":"Premise-Augmented Reasoning Chains Improve Error Identification in Math reasoning with LLMs","date":"2025-02-04","arxiv_id":"2502.02362","repositories_listed":0,"syntology":null},{"url":null,"slug":"smollm2-when-smol-goes-big-data-centric","title":"SmolLM2: When Smol Goes Big -- Data-Centric Training of a Small Language Model","date":"2025-02-04","arxiv_id":"2502.02737","repositories_listed":0,"syntology":null},{"url":null,"slug":"blink-of-an-eye-a-simple-theory-for-feature","title":"Blink of an eye: a simple theory for feature localization in generative models","date":"2025-02-02","arxiv_id":"2502.00921","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-autonomous-code-integration-for-math","title":"Learning Autonomous Code Integration for Math Language Models","date":"2025-02-02","arxiv_id":"2502.00691","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-mixture-of-agents-is-mixing","title":"Rethinking Mixture-of-Agents: Is Mixing Different Large Language Models Beneficial?","date":"2025-02-02","arxiv_id":"2502.00674","repositories_listed":0,"syntology":null},{"url":null,"slug":"brite-bootstrapping-reinforced-thinking","title":"BRiTE: Bootstrapping Reinforced Thinking Process to Enhance Language Model Reasoning","date":"2025-01-31","arxiv_id":"2501.18858","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairshare-data-pricing-for-large-language","title":"Fairshare Data Pricing via Data Valuation for Large Language Models","date":"2025-01-31","arxiv_id":"2502.00198","repositories_listed":0,"syntology":null},{"url":null,"slug":"pheromone-based-learning-of-optimal-reasoning","title":"Pheromone-based Learning of Optimal Reasoning Paths","date":"2025-01-31","arxiv_id":"2501.19278","repositories_listed":0,"syntology":null},{"url":null,"slug":"pixelworld-towards-perceiving-everything-as","title":"PixelWorld: Towards Perceiving Everything as Pixels","date":"2025-01-31","arxiv_id":"2501.19339","repositories_listed":0,"syntology":null},{"url":null,"slug":"examining-the-robustness-of-large-language","title":"Examining the Robustness of Large Language Models across Language Complexity","date":"2025-01-30","arxiv_id":"2501.18738","repositories_listed":0,"syntology":null},{"url":null,"slug":"token-hungry-yet-precise-deepseek-r1","title":"Token-Hungry, Yet Precise: DeepSeek R1 Highlights the Need for Multi-Step Reasoning Over Speed in MATH","date":"2025-01-30","arxiv_id":"2501.18576","repositories_listed":0,"syntology":null},{"url":null,"slug":"token-by-token-regeneration-and-domain-biases","title":"Token-by-Token Regeneration and Domain Biases: A Benchmark of LLMs on Advanced Mathematical Problem-Solving","date":"2025-01-28","arxiv_id":"2501.17084","repositories_listed":0,"syntology":null},{"url":null,"slug":"error-classification-of-large-language-models","title":"Error Classification of Large Language Models on Math Word Problems: A Dynamically Adaptive Framework","date":"2025-01-26","arxiv_id":"2501.15581","repositories_listed":0,"syntology":null}],"record_sha256":"9ce37aaffba456c2ea89d04dfd0c943c7271afb4077b69da7e5a56459da6ff96","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}