{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mathematical-reasoning/papers/5","list_of":"/task/mathematical-reasoning","task":"Mathematical Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":9,"rows_per_page":100,"rows":[401,500],"of":805,"counts":{"archive_papers_tagged":805,"with_a_code_link":395,"where_syntology_ran_a_sample":197,"not_listed_spam_title":0,"listed":805,"listed_where_code_ran":197,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":159,"every_run_a_failure_of_syntologys_instrument":38,"listed_with_a_run_with_no_instrument_failure":159,"listed_every_run_a_failure_of_syntologys_instrument":38,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mathematical-reasoning","prev":"/task/mathematical-reasoning/papers/4","next":"/task/mathematical-reasoning/papers/6","papers":[{"url":null,"slug":"core-enhancing-metacognition-with-label-free","title":"CoRE: Enhancing Metacognition with Label-free Self-evaluation in LRMs","date":"2025-07-08","arxiv_id":"2507.06087","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-don-t-make-sense-of","title":"Large Language Models Don't Make Sense of Word Problems. A Scoping Review from a Mathematics Education Perspective","date":"2025-06-30","arxiv_id":"2506.24006","repositories_listed":0,"syntology":null},{"url":null,"slug":"layer-importance-for-mathematical-reasoning","title":"Layer Importance for Mathematical Reasoning is Forged in Pre-Training and Invariant after Post-Training","date":"2025-06-27","arxiv_id":"2506.22638","repositories_listed":0,"syntology":null},{"url":null,"slug":"inside-you-are-many-wolves-using-cognitive","title":"Inside you are many wolves: Using cognitive models to interpret value trade-offs in LLMs","date":"2025-06-25","arxiv_id":"2506.20666","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-time-scaling-techniques-in-theoretical","title":"Test-time Scaling Techniques in Theoretical Physics -- A Comparison of Methods on the TPBench Dataset","date":"2025-06-25","arxiv_id":"2506.20729","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapthink-adaptive-thinking-preferences-for","title":"AdapThink: Adaptive Thinking Preferences for Reasoning Language Model","date":"2025-06-23","arxiv_id":"2506.18237","repositories_listed":0,"syntology":null},{"url":null,"slug":"physunibench-an-undergraduate-level-physics","title":"PhysUniBench: An Undergraduate-Level Physics Reasoning Benchmark for Multimodal Models","date":"2025-06-21","arxiv_id":"2506.17667","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-advanced-mathematical-reasoning-for","title":"Towards Advanced Mathematical Reasoning for LLMs via First-Order Logic Theorem Proving","date":"2025-06-20","arxiv_id":"2506.17104","repositories_listed":0,"syntology":null},{"url":null,"slug":"massive-supervised-fine-tuning-experiments","title":"Massive Supervised Fine-tuning Experiments Reveal How Data, Layer, and Training Factors Shape LLM Alignment Quality","date":"2025-06-17","arxiv_id":"2506.14681","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-chain-of-thought-prompting-zero","title":"Revisiting Chain-of-Thought Prompting: Zero-shot Can Be Stronger than Few-shot","date":"2025-06-17","arxiv_id":"2506.14641","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-technical-study-into-small-reasoning","title":"A Technical Study into Small Reasoning Language Models","date":"2025-06-16","arxiv_id":"2506.13404","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-interaction-of-linguistic","title":"Investigating the interaction of linguistic and mathematical reasoning in language models using multilingual number puzzles","date":"2025-06-16","arxiv_id":"2506.13886","repositories_listed":0,"syntology":null},{"url":null,"slug":"eliciting-reasoning-in-language-models-with","title":"Eliciting Reasoning in Language Models with Cognitive Tools","date":"2025-06-13","arxiv_id":"2506.12115","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-potential-of-large-language","title":"Investigating the Potential of Large Language Model-Based Router Multi-Agent Architectures for Foundation Design Automation: A Task Classification and Expert Selection Study","date":"2025-06-13","arxiv_id":"2506.13811","repositories_listed":0,"syntology":null},{"url":null,"slug":"learnalign-reasoning-data-selection-for","title":"LearnAlign: Reasoning Data Selection for Reinforcement Learning in Large Language Models Based on Improved Gradient Alignment","date":"2025-06-13","arxiv_id":"2506.11480","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-gold-standards-epistemic-ensemble-of","title":"Beyond Gold Standards: Epistemic Ensemble of LLM Judges for Formal Mathematical Reasoning","date":"2025-06-12","arxiv_id":"2506.10903","repositories_listed":0,"syntology":null},{"url":null,"slug":"premise-scalable-and-strategic-prompt","title":"PREMISE: Scalable and Strategic Prompt Optimization for Efficient Mathematical Reasoning in Large Models","date":"2025-06-12","arxiv_id":"2506.10716","repositories_listed":0,"syntology":null},{"url":null,"slug":"slimming-down-llms-without-losing-their-minds","title":"Slimming Down LLMs Without Losing Their Minds","date":"2025-06-12","arxiv_id":"2506.10885","repositories_listed":0,"syntology":null},{"url":null,"slug":"telemath-a-benchmark-for-large-language","title":"TeleMath: A Benchmark for Large Language Models in Telecom Mathematical Problem Solving","date":"2025-06-12","arxiv_id":"2506.10674","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-for-design-structure","title":"Large Language Models for Design Structure Matrix Optimization","date":"2025-06-11","arxiv_id":"2506.09749","repositories_listed":0,"syntology":null},{"url":null,"slug":"omni-dpo-a-dual-perspective-paradigm-for","title":"Omni-DPO: A Dual-Perspective Paradigm for Dynamic Preference Learning of LLMs","date":"2025-06-11","arxiv_id":"2506.10054","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-and-effective-alignment-of","title":"Towards Efficient and Effective Alignment of Large Language Models","date":"2025-06-11","arxiv_id":"2506.09329","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-large-language-models-for-5","title":"A Survey on Large Language Models for Mathematical Reasoning","date":"2025-06-10","arxiv_id":"2506.08446","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-have-intrinsic-meta","title":"Large Language Models Have Intrinsic Meta-Cognition, but Need a Good Lens","date":"2025-06-10","arxiv_id":"2506.08410","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporalizing-confidence-evaluation-of-chain","title":"Temporalizing Confidence: Evaluation of Chain-of-Thought Reasoning with Signal Temporal Logic","date":"2025-06-09","arxiv_id":"2506.08243","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-theoretical-physics-research-benefit-from","title":"Can Theoretical Physics Research Benefit from Language Agents?","date":"2025-06-06","arxiv_id":"2506.06214","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-accuracy-dissecting-mathematical","title":"Beyond Accuracy: Dissecting Mathematical Reasoning for LLMs Under Reinforcement Learning","date":"2025-06-05","arxiv_id":"2506.04723","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-layer-grpo-enhancing-reasoning-and-self","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","date":"2025-06-05","arxiv_id":"2506.04746","repositories_listed":0,"syntology":null},{"url":null,"slug":"prorefine-inference-time-prompt-refinement","title":"ProRefine: Inference-time Prompt Refinement with Textual Feedback","date":"2025-06-05","arxiv_id":"2506.05305","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-test-time-scaling-a-survey-and-a","title":"Revisiting Test-Time Scaling: A Survey and a Diversity-Aware Method for Efficient Reasoning","date":"2025-06-05","arxiv_id":"2506.04611","repositories_listed":0,"syntology":null},{"url":null,"slug":"videomathqa-benchmarking-mathematical","title":"VideoMathQA: Benchmarking Mathematical Reasoning via Multimodal Understanding in Videos","date":"2025-06-05","arxiv_id":"2506.05349","repositories_listed":0,"syntology":null},{"url":null,"slug":"webchorearena-evaluating-web-browsing-agents","title":"WebChoreArena: Evaluating Web Browsing Agents on Realistic Tedious Web Tasks","date":"2025-06-02","arxiv_id":"2506.01952","repositories_listed":0,"syntology":null},{"url":null,"slug":"gthinker-towards-general-multimodal-reasoning","title":"GThinker: Towards General Multimodal Reasoning via Cue-Guided Rethinking","date":"2025-06-01","arxiv_id":"2506.01078","repositories_listed":0,"syntology":null},{"url":"/paper/uni-lora-one-vector-is-all-you-need","slug":"uni-lora-one-vector-is-all-you-need","title":"Uni-LoRA: One Vector is All You Need","date":"2025-06-01","arxiv_id":"2506.00799","repositories_listed":0,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/uni-lora-one-vector-is-all-you-need#ran","syntology_url":"https://syntology.ai/paper/2506.00799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00799"}},"official":null}},{"url":null,"slug":"evaluation-of-llms-for-mathematical-problem","title":"Evaluation of LLMs for mathematical problem solving","date":"2025-05-30","arxiv_id":"2506.00309","repositories_listed":0,"syntology":null},{"url":null,"slug":"autogps-automated-geometry-problem-solving","title":"AutoGPS: Automated Geometry Problem Solving via Multimodal Formalization and Deductive Reasoning","date":"2025-05-29","arxiv_id":"2505.23381","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversity-aware-policy-optimization-for-large","title":"Diversity-Aware Policy Optimization for Large Language Model Reasoning","date":"2025-05-29","arxiv_id":"2505.23433","repositories_listed":0,"syntology":null},{"url":null,"slug":"let-s-reason-formally-natural-formal-hybrid","title":"Let's Reason Formally: Natural-Formal Hybrid Reasoning Enhances LLM's Math Capability","date":"2025-05-29","arxiv_id":"2505.23703","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-multi-agent-debate-as-test-time","title":"Revisiting Multi-Agent Debate as Test-Time Scaling: A Systematic Study of Conditional Effectiveness","date":"2025-05-29","arxiv_id":"2505.22960","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-overthinking-in-long-chain-of","title":"Revisiting Overthinking in Long Chain-of-Thought from the Perspective of Self-Doubt","date":"2025-05-29","arxiv_id":"2505.23480","repositories_listed":0,"syntology":null},{"url":null,"slug":"don-t-think-longer-think-wisely-optimizing","title":"Don't Think Longer, Think Wisely: Optimizing Thinking Dynamics for Large Reasoning Models","date":"2025-05-27","arxiv_id":"2505.21765","repositories_listed":0,"syntology":null},{"url":null,"slug":"enigmata-scaling-logical-reasoning-in-large","title":"Enigmata: Scaling Logical Reasoning in Large Language Models with Synthetic Verifiable Puzzles","date":"2025-05-26","arxiv_id":"2505.19914","repositories_listed":0,"syntology":null},{"url":null,"slug":"hs-star-hierarchical-sampling-for-self-taught","title":"HS-STAR: Hierarchical Sampling for Self-Taught Reasoners via Difficulty Estimation and Budget Reallocation","date":"2025-05-26","arxiv_id":"2505.19866","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-multilingual-math-reasoning-for","title":"Improving Multilingual Math Reasoning for African Languages","date":"2025-05-26","arxiv_id":"2505.19848","repositories_listed":0,"syntology":null},{"url":null,"slug":"activedpo-active-direct-preference","title":"ActiveDPO: Active Direct Preference Optimization for Sample-Efficient Alignment","date":"2025-05-25","arxiv_id":"2505.19241","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai4math-a-native-spanish-benchmark-for","title":"AI4Math: A Native Spanish Benchmark for University-Level Mathematical Reasoning in Large Language Models","date":"2025-05-25","arxiv_id":"2505.18978","repositories_listed":0,"syntology":null},{"url":null,"slug":"don-t-look-only-once-towards-multimodal","title":"Don't Look Only Once: Towards Multimodal Interactive Reasoning with Selective Visual Revisitation","date":"2025-05-24","arxiv_id":"2505.18842","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-long-cot-reasoning-in-small","title":"Efficient Long CoT Reasoning in Small Language Models","date":"2025-05-24","arxiv_id":"2505.18440","repositories_listed":0,"syntology":null},{"url":null,"slug":"logiccat-a-chain-of-thought-text-to-sql","title":"LogicCat: A Chain-of-Thought Text-to-SQL Benchmark for Multi-Domain Reasoning Challenges","date":"2025-05-24","arxiv_id":"2505.18744","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-by-gut-efficient-test-time-scaling","title":"Guided by Gut: Efficient Test-Time Scaling with Reinforced Intrinsic Confidence","date":"2025-05-23","arxiv_id":"2505.20325","repositories_listed":0,"syntology":null},{"url":null,"slug":"ppt-a-process-based-preference-learning","title":"PPT: A Process-based Preference Learning Framework for Self Improving Table Question Answering Models","date":"2025-05-23","arxiv_id":"2505.17565","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-unreasonable-effectiveness-of-model","title":"The Unreasonable Effectiveness of Model Merging for Cross-Lingual Transfer in LLMs","date":"2025-05-23","arxiv_id":"2505.18356","repositories_listed":0,"syntology":null},{"url":null,"slug":"amplify-adjacent-token-differences-enhancing","title":"Amplify Adjacent Token Differences: Enhancing Long Chain-of-Thought Reasoning with Shift-FFN","date":"2025-05-22","arxiv_id":"2505.17153","repositories_listed":0,"syntology":null},{"url":null,"slug":"bottlenecked-transformers-periodic-kv-cache","title":"Bottlenecked Transformers: Periodic KV Cache Abstraction for Generalised Reasoning","date":"2025-05-22","arxiv_id":"2505.16950","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-sampling-that-adapts-iterative-dpo","title":"Dynamic Sampling that Adapts: Iterative DPO for Self-Aware Mathematical Reasoning","date":"2025-05-22","arxiv_id":"2505.16176","repositories_listed":0,"syntology":null},{"url":null,"slug":"hoft-householder-orthogonal-fine-tuning","title":"HOFT: Householder Orthogonal Fine-tuning","date":"2025-05-22","arxiv_id":"2505.16531","repositories_listed":0,"syntology":null},{"url":null,"slug":"mcp-radar-a-multi-dimensional-benchmark-for","title":"MCP-RADAR: A Multi-Dimensional Benchmark for Evaluating Tool Use Capabilities in Large Language Models","date":"2025-05-22","arxiv_id":"2505.16700","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-self-generating-and-self-validating","title":"SMART: Self-Generating and Self-Validating Multi-Dimensional Assessment for LLMs' Mathematical Problem Solving","date":"2025-05-22","arxiv_id":"2505.16646","repositories_listed":0,"syntology":null},{"url":null,"slug":"think-silently-think-fast-dynamic-latent","title":"Think Silently, Think Fast: Dynamic Latent Compression of LLM Reasoning Chains","date":"2025-05-22","arxiv_id":"2505.16552","repositories_listed":0,"syntology":null},{"url":"/paper/tropical-attention-neural-algorithmic","slug":"tropical-attention-neural-algorithmic","title":"Tropical Attention: Neural Algorithmic Reasoning for Combinatorial Algorithms","date":"2025-05-22","arxiv_id":"2505.17190","repositories_listed":0,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tropical-attention-neural-algorithmic#ran","syntology_url":"https://syntology.ai/paper/2505.17190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17190"}},"official":null}},{"url":null,"slug":"can-llms-textit-understand-math-exploring-the","title":"Can LLMs $\\textit{understand}$ Math? -- Exploring the Pitfalls in Mathematical Reasoning","date":"2025-05-21","arxiv_id":"2505.15623","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-rank-chain-of-thought-an-energy","title":"Learning to Rank Chain-of-Thought: An Energy-Based Approach with Outcome Supervision","date":"2025-05-21","arxiv_id":"2505.14999","repositories_listed":0,"syntology":null},{"url":null,"slug":"maps-a-multilingual-benchmark-for-global","title":"MAPS: A Multilingual Benchmark for Global Agent Performance and Security","date":"2025-05-21","arxiv_id":"2505.15935","repositories_listed":0,"syntology":null},{"url":null,"slug":"ssr-speculative-parallel-scaling-reasoning-in","title":"SSR: Speculative Parallel Scaling Reasoning in Test-time","date":"2025-05-21","arxiv_id":"2505.15340","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-spoken-mathematical-reasoning","title":"Towards Spoken Mathematical Reasoning: Benchmarking Speech-based Models over Multi-faceted Math Problems","date":"2025-05-21","arxiv_id":"2505.15000","repositories_listed":0,"syntology":null},{"url":"/paper/trajectory-bellman-residual-minimization-a","slug":"trajectory-bellman-residual-minimization-a","title":"Trajectory Bellman Residual Minimization: A Simple Value-Based Method for LLM Reasoning","date":"2025-05-21","arxiv_id":"2505.15311","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trajectory-bellman-residual-minimization-a#ran","syntology_url":"https://syntology.ai/paper/2505.15311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15311"}},"official":null}},{"url":null,"slug":"aapo-enhance-the-reasoning-capabilities-of","title":"AAPO: Enhance the Reasoning Capabilities of LLMs with Advantage Momentum","date":"2025-05-20","arxiv_id":"2505.14264","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-the-first-error-process-reward-models","title":"Beyond the First Error: Process Reward Models for Reflective Mathematical Reasoning","date":"2025-05-20","arxiv_id":"2505.14391","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-pruning-improve-reasoning-revisiting-long","title":"Can Pruning Improve Reasoning? Revisiting Long-CoT Compression with Capability in Mind for Better Reasoning","date":"2025-05-20","arxiv_id":"2505.14582","repositories_listed":0,"syntology":null},{"url":null,"slug":"drp-distilled-reasoning-pruning-with-skill","title":"DRP: Distilled Reasoning Pruning with Skill-aware Step Decomposition for Efficient Large Reasoning Models","date":"2025-05-20","arxiv_id":"2505.13975","repositories_listed":0,"syntology":null},{"url":null,"slug":"mind-the-gap-bridging-thought-leap-for","title":"Mind the Gap: Bridging Thought Leap for Improved Chain-of-Thought Tuning","date":"2025-05-20","arxiv_id":"2505.14684","repositories_listed":0,"syntology":null},{"url":null,"slug":"osora-output-dimension-and-singular-value","title":"OSoRA: Output-Dimension and Singular-Value Initialized Low-Rank Adaptation","date":"2025-05-20","arxiv_id":"2505.14350","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-generation-beyond-discrete-token","title":"Text Generation Beyond Discrete Token Sampling","date":"2025-05-20","arxiv_id":"2505.14827","repositories_listed":0,"syntology":null},{"url":null,"slug":"wirelessmathbench-a-mathematical-modeling","title":"WirelessMathBench: A Mathematical Modeling Benchmark for LLMs in Wireless Communications","date":"2025-05-20","arxiv_id":"2505.14354","repositories_listed":0,"syntology":null},{"url":null,"slug":"automathkg-the-automated-mathematical","title":"AutoMathKG: The automated mathematical knowledge graph based on LLM and vector database","date":"2025-05-19","arxiv_id":"2505.13406","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-head-gating-a-framework-for","title":"Causal Head Gating: A Framework for Interpreting Roles of Attention Heads in Transformers","date":"2025-05-19","arxiv_id":"2505.13737","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-search-strategies-in-non-serializable","title":"Guided Search Strategies in Non-Serializable Environments with Applications to Software Engineering Agents","date":"2025-05-19","arxiv_id":"2505.13652","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-code-generation-for-functional","title":"Selective Code Generation for Functional Guarantees","date":"2025-05-19","arxiv_id":"2505.13553","repositories_listed":0,"syntology":null},{"url":null,"slug":"step-wise-adaptive-integration-of-supervised","title":"Step-wise Adaptive Integration of Supervised Fine-tuning and Reinforcement Learning for Task-Specific LLMs","date":"2025-05-19","arxiv_id":"2505.13026","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlocking-the-potential-of-difficulty-prior","title":"Unlocking the Potential of Difficulty Prior in RL-based Multimodal Reasoning","date":"2025-05-19","arxiv_id":"2505.13261","repositories_listed":0,"syntology":null},{"url":null,"slug":"seed-grpo-semantic-entropy-enhanced-grpo-for","title":"SEED-GRPO: Semantic Entropy Enhanced GRPO for Uncertainty-Aware Policy Optimization","date":"2025-05-18","arxiv_id":"2505.12346","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-verification-of-embodied-reasoning","title":"Real-Time Verification of Embodied Reasoning for Generative Skill Acquisition","date":"2025-05-16","arxiv_id":"2505.11175","repositories_listed":0,"syntology":null},{"url":"/paper/token-level-uncertainty-estimation-for-large","slug":"token-level-uncertainty-estimation-for-large","title":"Token-Level Uncertainty Estimation for Large Language Model Reasoning","date":"2025-05-16","arxiv_id":"2505.11737","repositories_listed":0,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/token-level-uncertainty-estimation-for-large#ran","syntology_url":"https://syntology.ai/paper/2505.11737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11737"}},"official":null}},{"url":null,"slug":"are-large-language-models-robust-in","title":"Are Large Language Models Robust in Understanding Code Against Semantics-Preserving Mutations?","date":"2025-05-15","arxiv_id":"2505.10443","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-as-a-service-based-on-agent-network","title":"Agent-as-a-Service based on Agent Network","date":"2025-05-13","arxiv_id":"2505.08446","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-like-humans-advancing-llm-reasoning","title":"Learning Like Humans: Advancing LLM Reasoning Capabilities via Adaptive Difficulty Curriculum Learning and Expert-Guided Self-Reformulation","date":"2025-05-13","arxiv_id":"2505.08364","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-robustness-to-spurious-correlations","title":"Assessing Robustness to Spurious Correlations in Post-Training Language Models","date":"2025-05-09","arxiv_id":"2505.05704","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-augmented-complex-problem-solving","title":"Knowledge Augmented Complex Problem Solving with Large Language Models: A Survey","date":"2025-05-06","arxiv_id":"2505.03418","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-the-last-answer-your-reasoning-trace","title":"Beyond the Last Answer: Your Reasoning Trace Uncovers More than You Think","date":"2025-04-29","arxiv_id":"2504.20708","repositories_listed":0,"syntology":null},{"url":null,"slug":"rv-syn-rational-and-verifiable-mathematical","title":"RV-Syn: Rational and Verifiable Mathematical Reasoning Data Synthesis based on Structured Function Library","date":"2025-04-29","arxiv_id":"2504.20426","repositories_listed":0,"syntology":null},{"url":null,"slug":"accurate-and-diverse-llm-mathematical","title":"Accurate and Diverse LLM Mathematical Reasoning via Automated PRM-Guided GFlowNets","date":"2025-04-28","arxiv_id":"2504.19981","repositories_listed":0,"syntology":null},{"url":null,"slug":"agentic-reasoning-and-tool-integration-for","title":"Agentic Reasoning and Tool Integration for LLMs via Reinforcement Learning","date":"2025-04-28","arxiv_id":"2505.01441","repositories_listed":0,"syntology":null},{"url":null,"slug":"spc-evolving-self-play-critic-via-adversarial","title":"SPC: Evolving Self-Play Critic via Adversarial Games for LLM Reasoning","date":"2025-04-27","arxiv_id":"2504.19162","repositories_listed":0,"syntology":null},{"url":"/paper/polymath-evaluating-mathematical-reasoning-in","slug":"polymath-evaluating-mathematical-reasoning-in","title":"PolyMath: Evaluating Mathematical Reasoning in Multilingual Contexts","date":"2025-04-25","arxiv_id":"2504.18428","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/polymath-evaluating-mathematical-reasoning-in#ran","syntology_url":"https://syntology.ai/paper/2504.18428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.18428"}},"official":null}},{"url":null,"slug":"deepdistill-enhancing-llm-reasoning","title":"DeepDistill: Enhancing LLM Reasoning Capabilities via Large-Scale Difficulty-Graded Data Training","date":"2025-04-24","arxiv_id":"2504.17565","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-grounded-reasoning-by-code","title":"Evaluating Grounded Reasoning by Code-Assisted Large Language Models for Mathematics","date":"2025-04-24","arxiv_id":"2504.17665","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-efficient-checkpoint-merging-via","title":"Parameter-Efficient Checkpoint Merging via Metrics-Weighted Averaging","date":"2025-04-23","arxiv_id":"2504.18580","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rl-exploration-for-llm-reasoning","title":"Improving RL Exploration for LLM Reasoning through Retrospective Replay","date":"2025-04-19","arxiv_id":"2504.14363","repositories_listed":0,"syntology":null},{"url":null,"slug":"bitnet-b1-58-2b4t-technical-report","title":"BitNet b1.58 2B4T Technical Report","date":"2025-04-16","arxiv_id":"2504.12285","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathematical-capabilities-of-large-language","title":"Assessment of Evolving Large Language Models in Upper Secondary Mathematics","date":"2025-04-15","arxiv_id":"2504.12347","repositories_listed":0,"syntology":null}],"record_sha256":"47b7821e2694ae5298d8486584332ca93318919d86f60958f23c89321a7384a3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}