{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/math/papers/9","list_of":"/task/math","task":"Math","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":16,"rows_per_page":100,"rows":[801,900],"of":1596,"counts":{"archive_papers_tagged":1596,"with_a_code_link":765,"where_syntology_ran_a_sample":349,"not_listed_spam_title":0,"listed":1596,"listed_where_code_ran":349,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/math","prev":"/task/math/papers/8","next":"/task/math/papers/10","papers":[{"url":null,"slug":"automatic-robustness-stress-testing-of-llms","title":"Automatic Robustness Stress Testing of LLMs as Mathematical Problem Solvers","date":"2025-06-05","arxiv_id":"2506.05038","repositories_listed":0,"syntology":null},{"url":null,"slug":"confidence-is-all-you-need-few-shot-rl-fine","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","date":"2025-06-05","arxiv_id":"2506.06395","repositories_listed":0,"syntology":null},{"url":null,"slug":"perceptual-decoupling-for-scalable-multi","title":"Perceptual Decoupling for Scalable Multi-modal Reasoning via Reward-Optimized Captioning","date":"2025-06-05","arxiv_id":"2506.04559","repositories_listed":0,"syntology":null},{"url":null,"slug":"simulating-llm-to-llm-tutoring-for","title":"Simulating LLM-to-LLM Tutoring for Multilingual Math Feedback","date":"2025-06-05","arxiv_id":"2506.04920","repositories_listed":0,"syntology":null},{"url":null,"slug":"treerpo-tree-relative-policy-optimization","title":"TreeRPO: Tree Relative Policy Optimization","date":"2025-06-05","arxiv_id":"2506.05183","repositories_listed":0,"syntology":null},{"url":null,"slug":"rectified-sparse-attention","title":"Rectified Sparse Attention","date":"2025-06-04","arxiv_id":"2506.04108","repositories_listed":0,"syntology":null},{"url":null,"slug":"master-enhancing-large-language-model-via","title":"MASTER: Enhancing Large Language Model via Multi-Agent Simulated Teaching","date":"2025-06-03","arxiv_id":"2506.02689","repositories_listed":0,"syntology":null},{"url":null,"slug":"unleashing-the-reasoning-potential-of-pre","title":"Unleashing the Reasoning Potential of Pre-trained LLMs by Critique Fine-Tuning on One Problem","date":"2025-06-03","arxiv_id":"2506.03295","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-or-reasoning-a-close-look-at-how","title":"Knowledge or Reasoning? A Close Look at How LLMs Think Across Domains","date":"2025-06-02","arxiv_id":"2506.02126","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthrl-scaling-visual-reasoning-with","title":"SynthRL: Scaling Visual Reasoning with Verifiable Data Synthesis","date":"2025-06-02","arxiv_id":"2506.02096","repositories_listed":0,"syntology":null},{"url":null,"slug":"gthinker-towards-general-multimodal-reasoning","title":"GThinker: Towards General Multimodal Reasoning via Cue-Guided Rethinking","date":"2025-06-01","arxiv_id":"2506.01078","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerated-sampling-from-masked-diffusion","title":"Accelerated Sampling from Masked Diffusion Models via Entropy Bounded Unmasking","date":"2025-05-30","arxiv_id":"2505.24857","repositories_listed":0,"syntology":null},{"url":null,"slug":"reflect-retry-reward-self-improving-llms-via","title":"Reflect, Retry, Reward: Self-Improving LLMs via Reinforcement Learning","date":"2025-05-30","arxiv_id":"2505.24726","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-llms-reason-abstractly-over-math-word","title":"Can LLMs Reason Abstractly Over Math Word Problems Without CoT? Disentangling Abstract Formulation From Arithmetic Computation","date":"2025-05-29","arxiv_id":"2505.23701","repositories_listed":0,"syntology":null},{"url":null,"slug":"dingo-constrained-inference-for-diffusion","title":"DINGO: Constrained Inference for Diffusion LLMs","date":"2025-05-29","arxiv_id":"2505.23061","repositories_listed":0,"syntology":null},{"url":null,"slug":"infi-mmr-curriculum-based-unlocking","title":"Infi-MMR: Curriculum-based Unlocking Multimodal Reasoning via Phased Reinforcement Learning in Multimodal Small Language Models","date":"2025-05-29","arxiv_id":"2505.23091","repositories_listed":0,"syntology":null},{"url":null,"slug":"let-s-reason-formally-natural-formal-hybrid","title":"Let's Reason Formally: Natural-Formal Hybrid Reasoning Enhances LLM's Math Capability","date":"2025-05-29","arxiv_id":"2505.23703","repositories_listed":0,"syntology":null},{"url":null,"slug":"matryoshka-model-learning-for-improved","title":"Matryoshka Model Learning for Improved Elastic Student Models","date":"2025-05-29","arxiv_id":"2505.23337","repositories_listed":0,"syntology":null},{"url":null,"slug":"pbebench-a-multi-step-programming-by-examples","title":"PBEBench: A Multi-Step Programming by Examples Reasoning Benchmark inspired by Historical Linguistics","date":"2025-05-29","arxiv_id":"2505.23126","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximizing-confidence-alone-improves","title":"Maximizing Confidence Alone Improves Reasoning","date":"2025-05-28","arxiv_id":"2505.22660","repositories_listed":0,"syntology":null},{"url":null,"slug":"walk-before-you-run-concise-llm-reasoning-via","title":"Walk Before You Run! Concise LLM Reasoning via Reinforcement Learning","date":"2025-05-27","arxiv_id":"2505.21178","repositories_listed":0,"syntology":null},{"url":null,"slug":"done-is-better-than-perfect-unlocking","title":"Done Is Better than Perfect: Unlocking Efficient Reasoning by Structured Multi-Turn Decomposition","date":"2025-05-26","arxiv_id":"2505.19788","repositories_listed":0,"syntology":null},{"url":null,"slug":"enigmata-scaling-logical-reasoning-in-large","title":"Enigmata: Scaling Logical Reasoning in Large Language Models with Synthetic Verifiable Puzzles","date":"2025-05-26","arxiv_id":"2505.19914","repositories_listed":0,"syntology":null},{"url":null,"slug":"faster-and-better-llms-via-latency-aware-test","title":"Faster and Better LLMs via Latency-Aware Test-Time Scaling","date":"2025-05-26","arxiv_id":"2505.19634","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-multilingual-math-reasoning-for","title":"Improving Multilingual Math Reasoning for African Languages","date":"2025-05-26","arxiv_id":"2505.19848","repositories_listed":0,"syntology":null},{"url":null,"slug":"interleaved-reasoning-for-large-language","title":"Interleaved Reasoning for Large Language Models via Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.19640","repositories_listed":0,"syntology":null},{"url":null,"slug":"prismatic-synthesis-gradient-based-data","title":"Prismatic Synthesis: Gradient-based Data Diversification Boosts Generalization in LLM Reasoning","date":"2025-05-26","arxiv_id":"2505.20161","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-role-of-diversity-in-in-context-learning","title":"The Role of Diversity in In-Context Learning for Large Language Models","date":"2025-05-26","arxiv_id":"2505.19426","repositories_listed":0,"syntology":null},{"url":null,"slug":"which-data-attributes-stimulate-math-and-code","title":"Which Data Attributes Stimulate Math and Code Reasoning? An Investigation via Influence Functions","date":"2025-05-26","arxiv_id":"2505.19949","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai4math-a-native-spanish-benchmark-for","title":"AI4Math: A Native Spanish Benchmark for University-Level Mathematical Reasoning in Large Language Models","date":"2025-05-25","arxiv_id":"2505.18978","repositories_listed":0,"syntology":null},{"url":null,"slug":"anchored-diffusion-language-model","title":"Anchored Diffusion Language Model","date":"2025-05-24","arxiv_id":"2505.18456","repositories_listed":0,"syntology":null},{"url":"/paper/does-representation-intervention-really","slug":"does-representation-intervention-really","title":"Does Representation Intervention Really Identify Desired Concepts and Elicit Alignment?","date":"2025-05-24","arxiv_id":"2505.18672","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/does-representation-intervention-really#ran","syntology_url":"https://syntology.ai/paper/2505.18672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18672"}},"official":null}},{"url":null,"slug":"msa-at-bea-2025-shared-task-disagreement","title":"MSA at BEA 2025 Shared Task: Disagreement-Aware Instruction Tuning for Multi-Dimensional Evaluation of LLMs as Math Tutors","date":"2025-05-24","arxiv_id":"2505.18549","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effect-of-negative-gradient-in-group","title":"On the Effect of Negative Gradient in Group Relative Deep Reinforcement Optimization","date":"2025-05-24","arxiv_id":"2505.18830","repositories_listed":0,"syntology":null},{"url":null,"slug":"steering-llm-reasoning-through-bias-only","title":"Steering LLM Reasoning Through Bias-Only Adaptation","date":"2025-05-24","arxiv_id":"2505.18706","repositories_listed":0,"syntology":null},{"url":null,"slug":"more-thinking-less-seeing-assessing-amplified","title":"More Thinking, Less Seeing? Assessing Amplified Hallucination in Multimodal Reasoning Models","date":"2025-05-23","arxiv_id":"2505.21523","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-rl-to-see-them-all-visual-triple-unified","title":"One RL to See Them All: Visual Triple Unified Reinforcement Learning","date":"2025-05-23","arxiv_id":"2505.18129","repositories_listed":0,"syntology":null},{"url":null,"slug":"outcome-based-reinforcement-learning-to","title":"Outcome-based Reinforcement Learning to Predict the Future","date":"2025-05-23","arxiv_id":"2505.17989","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-unreasonable-effectiveness-of-model","title":"The Unreasonable Effectiveness of Model Merging for Cross-Lingual Transfer in LLMs","date":"2025-05-23","arxiv_id":"2505.18356","repositories_listed":0,"syntology":null},{"url":null,"slug":"videogamebench-can-vision-language-models","title":"VideoGameBench: Can Vision-Language Models complete popular video games?","date":"2025-05-23","arxiv_id":"2505.18134","repositories_listed":0,"syntology":null},{"url":null,"slug":"acereason-nemotron-advancing-math-and-code","title":"AceReason-Nemotron: Advancing Math and Code Reasoning through Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.16400","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-sequence-classification-with","title":"Incremental Sequence Classification with Temporal Consistency","date":"2025-05-22","arxiv_id":"2505.16548","repositories_listed":0,"syntology":null},{"url":null,"slug":"rbench-v-a-primary-assessment-for-visual","title":"RBench-V: A Primary Assessment for Visual Reasoning Models with Multi-modal Outputs","date":"2025-05-22","arxiv_id":"2505.16770","repositories_listed":0,"syntology":null},{"url":null,"slug":"veracity-bias-and-beyond-uncovering-llms","title":"Veracity Bias and Beyond: Uncovering LLMs' Hidden Beliefs in Problem-Solving Reasoning","date":"2025-05-22","arxiv_id":"2505.16128","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-llms-textit-understand-math-exploring-the","title":"Can LLMs $\\textit{understand}$ Math? -- Exploring the Pitfalls in Mathematical Reasoning","date":"2025-05-21","arxiv_id":"2505.15623","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-rank-chain-of-thought-an-energy","title":"Learning to Rank Chain-of-Thought: An Energy-Based Approach with Outcome Supervision","date":"2025-05-21","arxiv_id":"2505.14999","repositories_listed":0,"syntology":null},{"url":null,"slug":"maps-a-multilingual-benchmark-for-global","title":"MAPS: A Multilingual Benchmark for Global Agent Performance and Security","date":"2025-05-21","arxiv_id":"2505.15935","repositories_listed":0,"syntology":null},{"url":null,"slug":"ssr-speculative-parallel-scaling-reasoning-in","title":"SSR: Speculative Parallel Scaling Reasoning in Test-time","date":"2025-05-21","arxiv_id":"2505.15340","repositories_listed":0,"syntology":null},{"url":null,"slug":"thought-augmented-policy-optimization","title":"Thought-Augmented Policy Optimization: Bridging External Guidance and Internal Capabilities","date":"2025-05-21","arxiv_id":"2505.15692","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-spoken-mathematical-reasoning","title":"Towards Spoken Mathematical Reasoning: Benchmarking Speech-based Models over Multi-faceted Math Problems","date":"2025-05-21","arxiv_id":"2505.15000","repositories_listed":0,"syntology":null},{"url":null,"slug":"easymath-a-0-shot-math-benchmark-for-slms","title":"EasyMath: A 0-shot Math Benchmark for SLMs","date":"2025-05-20","arxiv_id":"2505.14852","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-of-thoughts-navigating-llm-reasoning-with","title":"RL of Thoughts: Navigating LLM Reasoning with Inference-time Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14140","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-hallucination-tax-of-reinforcement","title":"The Hallucination Tax of Reinforcement Finetuning","date":"2025-05-20","arxiv_id":"2505.13988","repositories_listed":0,"syntology":null},{"url":null,"slug":"unearthing-gems-from-stones-policy","title":"Unearthing Gems from Stones: Policy Optimization with Negative Sample Augmentation for LLM Reasoning","date":"2025-05-20","arxiv_id":"2505.14403","repositories_listed":0,"syntology":null},{"url":null,"slug":"automathkg-the-automated-mathematical","title":"AutoMathKG: The automated mathematical knowledge graph based on LLM and vector database","date":"2025-05-19","arxiv_id":"2505.13406","repositories_listed":0,"syntology":null},{"url":null,"slug":"seed-grpo-semantic-entropy-enhanced-grpo-for","title":"SEED-GRPO: Semantic Entropy Enhanced GRPO for Uncertainty-Aware Policy Optimization","date":"2025-05-18","arxiv_id":"2505.12346","repositories_listed":0,"syntology":null},{"url":null,"slug":"lorasuite-efficient-lora-adaptation-across","title":"LoRASuite: Efficient LoRA Adaptation Across Large Language Model Upgrades","date":"2025-05-17","arxiv_id":"2505.13515","repositories_listed":0,"syntology":null},{"url":null,"slug":"mol-for-llms-dual-loss-optimization-to","title":"MoL for LLMs: Dual-Loss Optimization to Enhance Domain Expertise While Preserving General Capabilities","date":"2025-05-17","arxiv_id":"2505.12043","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11274","title":"SelfBudgeter: Adaptive Token Allocation for Efficient LLM Reasoning","date":"2025-05-16","arxiv_id":"2505.11274","repositories_listed":0,"syntology":null},{"url":null,"slug":"critique-guided-distillation-improving","title":"Critique-Guided Distillation: Improving Supervised Fine-tuning via Better Distillation","date":"2025-05-16","arxiv_id":"2505.11628","repositories_listed":0,"syntology":null},{"url":null,"slug":"dif-a-framework-for-benchmarking-and","title":"DIF: A Framework for Benchmarking and Verifying Implicit Bias in LLMs","date":"2025-05-15","arxiv_id":"2505.10013","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcing-the-diffusion-chain-of-lateral","title":"Reinforcing the Diffusion Chain of Lateral Thought with Diffusion Language Models","date":"2025-05-15","arxiv_id":"2505.10446","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-chain-of-thought-reasoning-when","title":"Accelerating Chain-of-Thought Reasoning: When Goal-Gradient Importance Meets Dynamic Skipping","date":"2025-05-13","arxiv_id":"2505.08392","repositories_listed":0,"syntology":null},{"url":null,"slug":"measurement-to-meaning-a-validity-centered","title":"Measurement to Meaning: A Validity-Centered Framework for AI Evaluation","date":"2025-05-13","arxiv_id":"2505.10573","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-peers-in-reasoning-models","title":"Learning from Peers in Reasoning Models","date":"2025-05-12","arxiv_id":"2505.07787","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-assessment-of-classroom-discourse","title":"Multimodal Assessment of Classroom Discourse Quality: A Text-Centered Attention-Based Multi-Task Learning Approach","date":"2025-05-12","arxiv_id":"2505.07902","repositories_listed":0,"syntology":null},{"url":null,"slug":"s-grpo-early-exit-via-reinforcement-learning","title":"S-GRPO: Early Exit via Reinforcement Learning in Reasoning Models","date":"2025-05-12","arxiv_id":"2505.07686","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialoguereason-rule-based-rl-sparks-dialogue","title":"DialogueReason: Rule-Based RL Sparks Dialogue Reasoning in LLMs","date":"2025-05-11","arxiv_id":"2505.07049","repositories_listed":0,"syntology":null},{"url":null,"slug":"xgen-small-technical-report","title":"xGen-small Technical Report","date":"2025-05-10","arxiv_id":"2505.06496","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-discovery-of-partial-differential","title":"Generative Discovery of Partial Differential Equations by Learning from Math Handbooks","date":"2025-05-09","arxiv_id":"2505.05869","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-llm-math-reasoning-acceleration-with","title":"Scalable LLM Math Reasoning Acceleration with Low-rank Distillation","date":"2025-05-08","arxiv_id":"2505.07861","repositories_listed":0,"syntology":null},{"url":null,"slug":"putting-the-value-back-in-rl-better-test-time","title":"Putting the Value Back in RL: Better Test-Time Scaling by Unifying LLM Reasoners With Verifiers","date":"2025-05-07","arxiv_id":"2505.04842","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-slow-thinking-based-reasoning","title":"A Survey of Slow Thinking-based Reasoning LLMs using Reinforced Learning and Inference-time Scaling Law","date":"2025-05-05","arxiv_id":"2505.02665","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-narrated-lecture-videos-from","title":"Generating Narrated Lecture Videos from Slides with Synchronized Highlights","date":"2025-05-05","arxiv_id":"2505.02966","repositories_listed":0,"syntology":null},{"url":null,"slug":"simplemix-frustratingly-simple-mixing-of-off","title":"SIMPLEMIX: Frustratingly Simple Mixing of Off- and On-policy Data in Language Model Preference Learning","date":"2025-05-05","arxiv_id":"2505.02363","repositories_listed":0,"syntology":null},{"url":null,"slug":"lookalike-consistent-distractor-generation-in","title":"LookAlike: Consistent Distractor Generation in Math MCQs","date":"2025-05-03","arxiv_id":"2505.01903","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptmi-adaptive-skill-based-in-context-math","title":"AdaptMI: Adaptive Skill-based In-context Math Instruction for Small Language Models","date":"2025-04-30","arxiv_id":"2505.00147","repositories_listed":0,"syntology":null},{"url":null,"slug":"llms-do-not-have-human-like-working-memory","title":"LLMs Do Not Have Human-Like Working Memory","date":"2025-04-30","arxiv_id":"2505.10571","repositories_listed":0,"syntology":null},{"url":null,"slug":"phi-4-mini-reasoning-exploring-the-limits-of","title":"Phi-4-Mini-Reasoning: Exploring the Limits of Small Reasoning Language Models in Math","date":"2025-04-30","arxiv_id":"2504.21233","repositories_listed":0,"syntology":null},{"url":null,"slug":"phi-4-reasoning-technical-report","title":"Phi-4-reasoning Technical Report","date":"2025-04-30","arxiv_id":"2504.21318","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-prompt-optimization","title":"Local Prompt Optimization","date":"2025-04-29","arxiv_id":"2504.20355","repositories_listed":0,"syntology":null},{"url":null,"slug":"trace-of-thought-prompting-investigating","title":"Trace-of-Thought Prompting: Investigating Prompt-Based Knowledge Distillation Through Question Decomposition","date":"2025-04-29","arxiv_id":"2504.20946","repositories_listed":0,"syntology":null},{"url":null,"slug":"accurate-and-diverse-llm-mathematical","title":"Accurate and Diverse LLM Mathematical Reasoning via Automated PRM-Guided GFlowNets","date":"2025-04-28","arxiv_id":"2504.19981","repositories_listed":0,"syntology":null},{"url":"/paper/ape-bench-i-towards-file-level-automated","slug":"ape-bench-i-towards-file-level-automated","title":"APE-Bench I: Towards File-level Automated Proof Engineering of Formal Math Libraries","date":"2025-04-27","arxiv_id":"2504.19110","repositories_listed":0,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/ape-bench-i-towards-file-level-automated#ran","syntology_url":"https://syntology.ai/paper/2504.19110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.19110"}},"official":null}},{"url":null,"slug":"evaluating-grounded-reasoning-by-code","title":"Evaluating Grounded Reasoning by Code-Assisted Large Language Models for Mathematics","date":"2025-04-24","arxiv_id":"2504.17665","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-large-language-models-to-reason-via","title":"Training Large Language Models to Reason via EM Policy Gradient","date":"2025-04-24","arxiv_id":"2504.18587","repositories_listed":0,"syntology":null},{"url":null,"slug":"splitreason-learning-to-offload-reasoning","title":"SplitReason: Learning To Offload Reasoning","date":"2025-04-23","arxiv_id":"2504.16379","repositories_listed":0,"syntology":null},{"url":null,"slug":"longperceptualthoughts-distilling-system-2","title":"LongPerceptualThoughts: Distilling System-2 Reasoning for System-1 Perception","date":"2025-04-21","arxiv_id":"2504.15362","repositories_listed":0,"syntology":null},{"url":null,"slug":"otc-optimal-tool-calls-via-reinforcement","title":"OTC: Optimal Tool Calls via Reinforcement Learning","date":"2025-04-21","arxiv_id":"2504.14870","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-reinforcement-learning-really","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","date":"2025-04-18","arxiv_id":"2504.13837","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-math-learning-in-an-lms-using-ai","title":"Enhancing Math Learning in an LMS Using AI-Driven Question Recommendations","date":"2025-04-18","arxiv_id":"2504.14098","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-between-myth-and-reality-ai-for-math-a","title":"In between myth and reality: AI for math -- a case study in category theory","date":"2025-04-17","arxiv_id":"2504.13360","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathphys-guided-coarse-to-fine-anomaly","title":"MathPhys-Guided Coarse-to-Fine Anomaly Synthesis with SQE-Driven Bi-Level Optimization for Anomaly Detection","date":"2025-04-17","arxiv_id":"2504.12970","repositories_listed":0,"syntology":null},{"url":null,"slug":"thoughtterminator-benchmarking-calibrating","title":"THOUGHTTERMINATOR: Benchmarking, Calibrating, and Mitigating Overthinking in Reasoning Models","date":"2025-04-17","arxiv_id":"2504.13367","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-guided-watermarking-for-llms-a-test","title":"Entropy-Guided Watermarking for LLMs: A Test-Time Framework for Robust and Traceable Text Generation","date":"2025-04-16","arxiv_id":"2504.12108","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-the-generation-of-high-quality-cot","title":"Rethinking the Generation of High-Quality CoT Data from the Perspective of LLM-Adaptive Question Difficulty Grading","date":"2025-04-16","arxiv_id":"2504.11919","repositories_listed":0,"syntology":null},{"url":null,"slug":"heimdall-test-time-scaling-on-the-generative","title":"Heimdall: test-time scaling on the generative verification","date":"2025-04-14","arxiv_id":"2504.10337","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpt-carry-on-training-foundation-model-for","title":"GPT Carry-On: Training Foundation Model for Customization Could Be Simple, Scalable and Affordable","date":"2025-04-10","arxiv_id":"2504.07513","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-optimism-correction-be-confident","title":"Supervised Optimism Correction: Be Confident When LLMs Are Sure","date":"2025-04-10","arxiv_id":"2504.07527","repositories_listed":0,"syntology":null},{"url":null,"slug":"mdit-a-model-free-data-interpolation-method","title":"MDIT: A Model-free Data Interpolation Method for Diverse Instruction Tuning","date":"2025-04-09","arxiv_id":"2504.07288","repositories_listed":0,"syntology":null}],"record_sha256":"fea770e9dd503475cf1f3b2d5c4bd808da6cd485f26e0a016189ec45496e7c74","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}