{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mathematical-reasoning/papers/2","list_of":"/task/mathematical-reasoning","task":"Mathematical Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":9,"rows_per_page":100,"rows":[101,200],"of":805,"counts":{"archive_papers_tagged":805,"with_a_code_link":395,"where_syntology_ran_a_sample":197,"not_listed_spam_title":0,"listed":805,"listed_where_code_ran":197,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":159,"every_run_a_failure_of_syntologys_instrument":38,"listed_with_a_run_with_no_instrument_failure":159,"listed_every_run_a_failure_of_syntologys_instrument":38,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mathematical-reasoning","prev":"/task/mathematical-reasoning","next":"/task/mathematical-reasoning/papers/3","papers":[{"url":"/paper/unraveling-misinformation-propagation-in-llm","slug":"unraveling-misinformation-propagation-in-llm","title":"Unraveling Misinformation Propagation in LLM Reasoning","date":"2025-05-24","arxiv_id":"2505.18555","repositories_listed":1,"syntology":null},{"url":"/paper/rader-reasoning-aware-dense-retrieval-models","slug":"rader-reasoning-aware-dense-retrieval-models","title":"RaDeR: Reasoning-aware Dense Retrieval Models","date":"2025-05-23","arxiv_id":"2505.18405","repositories_listed":1,"syntology":null},{"url":"/paper/equivpruner-boosting-efficiency-and-quality-1","slug":"equivpruner-boosting-efficiency-and-quality-1","title":"EquivPruner: Boosting Efficiency and Quality in LLM-Based Search via Action Pruning","date":"2025-05-22","arxiv_id":"2505.16312","repositories_listed":1,"syntology":null},{"url":"/paper/ktae-a-model-free-algorithm-to-key-tokens","slug":"ktae-a-model-free-algorithm-to-key-tokens","title":"KTAE: A Model-Free Algorithm to Key-Tokens Advantage Estimation in Mathematical Reasoning","date":"2025-05-22","arxiv_id":"2505.16826","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/ktae-a-model-free-algorithm-to-key-tokens#ran","syntology_url":"https://syntology.ai/paper/2505.16826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16826"}},"official":{"repos":["xiaolizh1/ktae"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/rl-tango-reinforcing-generator-and-verifier","slug":"rl-tango-reinforcing-generator-and-verifier","title":"RL Tango: Reinforcing Generator and Verifier Together for Language Reasoning","date":"2025-05-21","arxiv_id":"2505.15034","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rl-tango-reinforcing-generator-and-verifier#ran","syntology_url":"https://syntology.ai/paper/2505.15034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15034"}},"official":{"repos":["kaiwenzha/rl-tango"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepeyes-incentivizing-thinking-with-images","slug":"deepeyes-incentivizing-thinking-with-images","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14362","repositories_listed":1,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deepeyes-incentivizing-thinking-with-images#ran","syntology_url":"https://syntology.ai/paper/2505.14362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14362"}},"official":{"repos":["visual-agent/deepeyes"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/general-reasoner-advancing-llm-reasoning","slug":"general-reasoner-advancing-llm-reasoning","title":"General-Reasoner: Advancing LLM Reasoning Across All Domains","date":"2025-05-20","arxiv_id":"2505.14652","repositories_listed":1,"syntology":null},{"url":"/paper/let-s-verify-math-questions-step-by-step","slug":"let-s-verify-math-questions-step-by-step","title":"Let's Verify Math Questions Step by Step","date":"2025-05-20","arxiv_id":"2505.13903","repositories_listed":1,"syntology":null},{"url":"/paper/mm-agent-llm-as-agents-for-real-world","slug":"mm-agent-llm-as-agents-for-real-world","title":"MM-Agent: LLM as Agents for Real-world Mathematical Modeling Problem","date":"2025-05-20","arxiv_id":"2505.14148","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-reasoning-losing-control-evaluating","slug":"scaling-reasoning-losing-control-evaluating","title":"Scaling Reasoning, Losing Control: Evaluating Instruction Following in Large Reasoning Models","date":"2025-05-20","arxiv_id":"2505.14810","repositories_listed":1,"syntology":null},{"url":"/paper/scope-compress-mathematical-reasoning-steps","slug":"scope-compress-mathematical-reasoning-steps","title":"SCOPE: Compress Mathematical Reasoning Steps for Efficient Automated Process Annotation","date":"2025-05-20","arxiv_id":"2505.14419","repositories_listed":1,"syntology":null},{"url":"/paper/mindomni-unleashing-reasoning-generation-in","slug":"mindomni-unleashing-reasoning-generation-in","title":"MindOmni: Unleashing Reasoning Generation in Vision Language Models with RGPO","date":"2025-05-19","arxiv_id":"2505.13031","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-anytime-reasoning-via-budget","slug":"optimizing-anytime-reasoning-via-budget","title":"Optimizing Anytime Reasoning via Budget Relative Policy Optimization","date":"2025-05-19","arxiv_id":"2505.13438","repositories_listed":1,"syntology":null},{"url":"/paper/trust-but-verify-a-self-verification-approach","slug":"trust-but-verify-a-self-verification-approach","title":"Trust, But Verify: A Self-Verification Approach to Reinforcement Learning with Verifiable Rewards","date":"2025-05-19","arxiv_id":"2505.13445","repositories_listed":1,"syntology":null},{"url":"/paper/disco-reinforcing-large-reasoning-models-with","slug":"disco-reinforcing-large-reasoning-models-with","title":"DisCO: Reinforcing Large Reasoning Models with Discriminative Constrained Optimization","date":"2025-05-18","arxiv_id":"2505.12366","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/disco-reinforcing-large-reasoning-models-with#ran","syntology_url":"https://syntology.ai/paper/2505.12366","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12366"}},"official":{"repos":["optimization-ai/disco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/marge-improving-math-reasoning-for-llms-with","slug":"marge-improving-math-reasoning-for-llms-with","title":"MARGE: Improving Math Reasoning for LLMs with Guided Exploration","date":"2025-05-18","arxiv_id":"2505.12500","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/marge-improving-math-reasoning-for-llms-with#ran","syntology_url":"https://syntology.ai/paper/2505.12500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12500"}},"official":{"repos":["georgao35/marge"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","unlocated"]}}},{"url":"/paper/realmath-a-continuous-benchmark-for","slug":"realmath-a-continuous-benchmark-for","title":"RealMath: A Continuous Benchmark for Evaluating Language Models on Research-Level Mathematics","date":"2025-05-18","arxiv_id":"2505.12575","repositories_listed":1,"syntology":null},{"url":"/paper/hardmath2-a-benchmark-for-applied-mathematics","slug":"hardmath2-a-benchmark-for-applied-mathematics","title":"HARDMath2: A Benchmark for Applied Mathematics Built by Students as Part of a Graduate Class","date":"2025-05-17","arxiv_id":"2505.11774","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10978","slug":"2505-10978","title":"Group-in-Group Policy Optimization for LLM Agent Training","date":"2025-05-16","arxiv_id":"2505.10978","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11140","slug":"2505-11140","title":"Scaling Reasoning can Improve Factuality in Large Language Models","date":"2025-05-16","arxiv_id":"2505.11140","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-on-a-budget-miniaturizing-deepseek","slug":"reasoning-on-a-budget-miniaturizing-deepseek","title":"Reasoning on a Budget: Miniaturizing DeepSeek R1 with SFT-GRPO Alignment for Instruction-Tuned LLMs","date":"2025-05-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/complexformer-disruptively-advancing","slug":"complexformer-disruptively-advancing","title":"ComplexFormer: Disruptively Advancing Transformer Inference Ability via Head-Specific Complex Vector Attention","date":"2025-05-15","arxiv_id":"2505.10222","repositories_listed":1,"syntology":null},{"url":"/paper/mathcoder-vl-bridging-vision-and-code-for","slug":"mathcoder-vl-bridging-vision-and-code-for","title":"MathCoder-VL: Bridging Vision and Code for Enhanced Multimodal Mathematical Reasoning","date":"2025-05-15","arxiv_id":"2505.10557","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathcoder-vl-bridging-vision-and-code-for#ran","syntology_url":"https://syntology.ai/paper/2505.10557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10557"}},"official":{"repos":["mathllm/mathcoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dra-grpo-exploring-diversity-aware-reward","slug":"dra-grpo-exploring-diversity-aware-reward","title":"DRA-GRPO: Exploring Diversity-Aware Reward Adjustment for R1-Zero-Like Training of Large Language Models","date":"2025-05-14","arxiv_id":"2505.09655","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dra-grpo-exploring-diversity-aware-reward#ran","syntology_url":"https://syntology.ai/paper/2505.09655","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.09655"}},"official":{"repos":["xiwenc1/dra-grpo"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crosslingual-reasoning-through-test-time","slug":"crosslingual-reasoning-through-test-time","title":"Crosslingual Reasoning through Test-Time Scaling","date":"2025-05-08","arxiv_id":"2505.05408","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/crosslingual-reasoning-through-test-time#ran","syntology_url":"https://syntology.ai/paper/2505.05408","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05408"}},"official":{"repos":["BatsResearch/crosslingual-test-time-scaling"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/formalmath-benchmarking-formal-mathematical","slug":"formalmath-benchmarking-formal-mathematical","title":"FormalMATH: Benchmarking Formal Mathematical Reasoning of Large Language Models","date":"2025-05-05","arxiv_id":"2505.02735","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/formalmath-benchmarking-formal-mathematical#ran","syntology_url":"https://syntology.ai/paper/2505.02735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02735"}},"official":{"repos":["sphere-ai-lab/formalmath-bench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/optimizing-chain-of-thought-reasoners-via","slug":"optimizing-chain-of-thought-reasoners-via","title":"Optimizing Chain-of-Thought Reasoners via Gradient Variance Minimization in Rejection Sampling and RL","date":"2025-05-05","arxiv_id":"2505.02391","repositories_listed":1,"syntology":null},{"url":"/paper/rewriting-pre-training-data-boosts-llm","slug":"rewriting-pre-training-data-boosts-llm","title":"Rewriting Pre-Training Data Boosts LLM Performance in Math and Code","date":"2025-05-05","arxiv_id":"2505.02881","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rewriting-pre-training-data-boosts-llm#ran","syntology_url":"https://syntology.ai/paper/2505.02881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02881"}},"official":{"repos":["rioyokotalab/swallow-code-math"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepseek-prover-v2-advancing-formal","slug":"deepseek-prover-v2-advancing-formal","title":"DeepSeek-Prover-V2: Advancing Formal Mathematical Reasoning via Reinforcement Learning for Subgoal Decomposition","date":"2025-04-30","arxiv_id":"2504.21801","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-reasoning-in-large","slug":"reinforcement-learning-for-reasoning-in-large","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","date":"2025-04-29","arxiv_id":"2504.20571","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":9,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/reinforcement-learning-for-reasoning-in-large#ran","syntology_url":"https://syntology.ai/paper/2504.20571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20571"}},"official":{"repos":["ypwang61/one-shot-rlvr"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-evaluative-thinking-meta-policy","slug":"toward-evaluative-thinking-meta-policy","title":"Toward Evaluative Thinking: Meta Policy Optimization with Evolving Reward Models","date":"2025-04-28","arxiv_id":"2504.20157","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":15,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/toward-evaluative-thinking-meta-policy#ran","syntology_url":"https://syntology.ai/paper/2504.20157","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20157"}},"official":{"repos":["minnesotanlp/mpo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-attention-generates-better","slug":"hierarchical-attention-generates-better","title":"Hierarchical Attention Generates Better Proofs","date":"2025-04-27","arxiv_id":"2504.19188","repositories_listed":1,"syntology":null},{"url":"/paper/2504-18589","slug":"2504-18589","title":"Benchmarking Multimodal Mathematical Reasoning with Explicit Visual Dependency","date":"2025-04-24","arxiv_id":"2504.18589","repositories_listed":1,"syntology":null},{"url":"/paper/aimo-2-winning-solution-building-state-of-the","slug":"aimo-2-winning-solution-building-state-of-the","title":"AIMO-2 Winning Solution: Building State-of-the-Art Mathematical Reasoning Models with OpenMathReasoning dataset","date":"2025-04-23","arxiv_id":"2504.16891","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/aimo-2-winning-solution-building-state-of-the#ran","syntology_url":"https://syntology.ai/paper/2504.16891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16891"}},"official":null}},{"url":"/paper/enhancing-the-geometric-problem-solving","slug":"enhancing-the-geometric-problem-solving","title":"Enhancing the Geometric Problem-Solving Ability of Multimodal LLMs via Symbolic-Neural Integration","date":"2025-04-17","arxiv_id":"2504.12773","repositories_listed":1,"syntology":null},{"url":"/paper/climbing-the-ladder-of-reasoning-what-llms","slug":"climbing-the-ladder-of-reasoning-what-llms","title":"Climbing the Ladder of Reasoning: What LLMs Can-and Still Can't-Solve after SFT?","date":"2025-04-16","arxiv_id":"2504.11741","repositories_listed":1,"syntology":null},{"url":"/paper/a-dual-space-framework-for-general-knowledge","slug":"a-dual-space-framework-for-general-knowledge","title":"A Dual-Space Framework for General Knowledge Distillation of Large Language Models","date":"2025-04-15","arxiv_id":"2504.11426","repositories_listed":1,"syntology":null},{"url":"/paper/deepmath-103k-a-large-scale-challenging","slug":"deepmath-103k-a-large-scale-challenging","title":"DeepMath-103K: A Large-Scale, Challenging, Decontaminated, and Verifiable Mathematical Dataset for Advancing Reasoning","date":"2025-04-15","arxiv_id":"2504.11456","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepmath-103k-a-large-scale-challenging#ran","syntology_url":"https://syntology.ai/paper/2504.11456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11456"}},"official":{"repos":["zwhe99/deepmath"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/retool-reinforcement-learning-for-strategic","slug":"retool-reinforcement-learning-for-strategic","title":"ReTool: Reinforcement Learning for Strategic Tool Use in LLMs","date":"2025-04-15","arxiv_id":"2504.11536","repositories_listed":1,"syntology":null},{"url":"/paper/teaching-large-language-models-to-reason","slug":"teaching-large-language-models-to-reason","title":"Teaching Large Language Models to Reason through Learning and Forgetting","date":"2025-04-15","arxiv_id":"2504.11364","repositories_listed":1,"syntology":null},{"url":"/paper/breaking-the-data-barrier-building-gui-agents","slug":"breaking-the-data-barrier-building-gui-agents","title":"Breaking the Data Barrier -- Building GUI Agents Through Task Generalization","date":"2025-04-14","arxiv_id":"2504.10127","repositories_listed":1,"syntology":null},{"url":"/paper/two-heads-are-better-than-one-test-time-1","slug":"two-heads-are-better-than-one-test-time-1","title":"Two Heads are Better Than One: Test-time Scaling of Multi-agent Collaborative Reasoning","date":"2025-04-14","arxiv_id":"2504.09772","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/two-heads-are-better-than-one-test-time-1#ran","syntology_url":"https://syntology.ai/paper/2504.09772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.09772"}},"official":{"repos":["jincan333/MAS-TTS"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grpo-lead-a-difficulty-aware-reinforcement-1","slug":"grpo-lead-a-difficulty-aware-reinforcement-1","title":"GRPO-LEAD: A Difficulty-Aware Reinforcement Learning Approach for Concise Mathematical Reasoning in Language Models","date":"2025-04-13","arxiv_id":"2504.09696","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/grpo-lead-a-difficulty-aware-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2504.09696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.09696"}},"official":{"repos":["aeroplanepaper/GRPO-LEAD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/echo-chamber-rl-post-training-amplifies","slug":"echo-chamber-rl-post-training-amplifies","title":"Echo Chamber: RL Post-training Amplifies Behaviors Learned in Pretraining","date":"2025-04-10","arxiv_id":"2504.07912","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/echo-chamber-rl-post-training-amplifies#ran","syntology_url":"https://syntology.ai/paper/2504.07912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07912"}},"official":{"repos":["rosieyzh/openrlhf-pretrain"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/kimi-vl-technical-report","slug":"kimi-vl-technical-report","title":"Kimi-VL Technical Report","date":"2025-04-10","arxiv_id":"2504.07491","repositories_listed":1,"syntology":null},{"url":"/paper/lori-reducing-cross-task-interference-in","slug":"lori-reducing-cross-task-interference-in","title":"LoRI: Reducing Cross-Task Interference in Multi-Task Low-Rank Adaptation","date":"2025-04-10","arxiv_id":"2504.07448","repositories_listed":1,"syntology":null},{"url":"/paper/alice-proactive-learning-with-teacher-s","slug":"alice-proactive-learning-with-teacher-s","title":"Alice: Proactive Learning with Teacher's Demonstrations for Weak-to-Strong Generalization","date":"2025-04-09","arxiv_id":"2504.07316","repositories_listed":1,"syntology":null},{"url":"/paper/right-question-is-already-half-the-answer","slug":"right-question-is-already-half-the-answer","title":"Right Question is Already Half the Answer: Fully Unsupervised LLM Reasoning Incentivization","date":"2025-04-08","arxiv_id":"2504.05812","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/right-question-is-already-half-the-answer#ran","syntology_url":"https://syntology.ai/paper/2504.05812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05812"}},"official":{"repos":["qingyangzhang/empo"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/efficient-reinforcement-finetuning-via","slug":"efficient-reinforcement-finetuning-via","title":"Efficient Reinforcement Finetuning via Adaptive Curriculum Learning","date":"2025-04-07","arxiv_id":"2504.05520","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/efficient-reinforcement-finetuning-via#ran","syntology_url":"https://syntology.ai/paper/2504.05520","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05520"}},"official":{"repos":["uscnlp-lime/verl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/do-llm-evaluators-prefer-themselves-for-a","slug":"do-llm-evaluators-prefer-themselves-for-a","title":"Do LLM Evaluators Prefer Themselves for a Reason?","date":"2025-04-04","arxiv_id":"2504.03846","repositories_listed":1,"syntology":null},{"url":"/paper/megamath-pushing-the-limits-of-open-math","slug":"megamath-pushing-the-limits-of-open-math","title":"MegaMath: Pushing the Limits of Open Math Corpora","date":"2025-04-03","arxiv_id":"2504.02807","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/megamath-pushing-the-limits-of-open-math#ran","syntology_url":"https://syntology.ai/paper/2504.02807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02807"}},"official":{"repos":["llm360/megamath"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/medreason-eliciting-factual-medical-reasoning","slug":"medreason-eliciting-factual-medical-reasoning","title":"MedReason: Eliciting Factual Medical Reasoning Steps in LLMs via Knowledge Graphs","date":"2025-04-01","arxiv_id":"2504.00993","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-mllm-reasoning-with-text-debiased","slug":"boosting-mllm-reasoning-with-text-debiased","title":"Boosting MLLM Reasoning with Text-Debiased Hint-GRPO","date":"2025-03-31","arxiv_id":"2503.23905","repositories_listed":1,"syntology":null},{"url":"/paper/embodied-reasoner-synergizing-visual-search","slug":"embodied-reasoner-synergizing-visual-search","title":"Embodied-Reasoner: Synergizing Visual Search, Reasoning, and Action for Embodied Interactive Tasks","date":"2025-03-27","arxiv_id":"2503.21696","repositories_listed":1,"syntology":null},{"url":"/paper/r-prm-reasoning-driven-process-reward","slug":"r-prm-reasoning-driven-process-reward","title":"R-PRM: Reasoning-Driven Process Reward Modeling","date":"2025-03-27","arxiv_id":"2503.21295","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r-prm-reasoning-driven-process-reward#ran","syntology_url":"https://syntology.ai/paper/2503.21295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21295"}},"official":{"repos":["njunlp/r-prm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/swi-speaking-with-intent-in-large-language","slug":"swi-speaking-with-intent-in-large-language","title":"SWI: Speaking with Intent in Large Language Models","date":"2025-03-27","arxiv_id":"2503.21544","repositories_listed":1,"syntology":null},{"url":"/paper/accelerate-parallelizable-reasoning-via","slug":"accelerate-parallelizable-reasoning-via","title":"Accelerate Parallelizable Reasoning via Parallel Decoding within One Sequence","date":"2025-03-26","arxiv_id":"2503.20533","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":5,"n_ran_checked":5,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"9 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/accelerate-parallelizable-reasoning-via#ran","syntology_url":"https://syntology.ai/paper/2503.20533","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20533"}},"official":{"repos":["yuyijiong/parallel-decoding-in-one-sequence"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/trajectory-balance-with-asynchrony-decoupling","slug":"trajectory-balance-with-asynchrony-decoupling","title":"Trajectory Balance with Asynchrony: Decoupling Exploration and Learning for Fast, Scalable LLM Post-Training","date":"2025-03-24","arxiv_id":"2503.18929","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/trajectory-balance-with-asynchrony-decoupling#ran","syntology_url":"https://syntology.ai/paper/2503.18929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18929"}},"official":null}},{"url":"/paper/lost-in-cultural-translation-do-llms-struggle","slug":"lost-in-cultural-translation-do-llms-struggle","title":"Lost in Cultural Translation: Do LLMs Struggle with Math Across Cultural Contexts?","date":"2025-03-23","arxiv_id":"2503.18018","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-mathematical-reasoning-and","slug":"a-survey-on-mathematical-reasoning-and","title":"A Survey on Mathematical Reasoning and Optimization with Large Language Models","date":"2025-03-22","arxiv_id":"2503.17726","repositories_listed":1,"syntology":null},{"url":"/paper/mathfusion-enhancing-mathematic-problem","slug":"mathfusion-enhancing-mathematic-problem","title":"MathFusion: Enhancing Mathematic Problem-solving of LLM through Instruction Fusion","date":"2025-03-20","arxiv_id":"2503.16212","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mathfusion-enhancing-mathematic-problem#ran","syntology_url":"https://syntology.ai/paper/2503.16212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16212"}},"official":{"repos":["qizhipei/mathfusion"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-reasoning-in-small","slug":"reinforcement-learning-for-reasoning-in-small","title":"Reinforcement Learning for Reasoning in Small LLMs: What Works and What Doesn't","date":"2025-03-20","arxiv_id":"2503.16219","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-for-reasoning-in-small#ran","syntology_url":"https://syntology.ai/paper/2503.16219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16219"}},"official":{"repos":["knoveleng/open-rs"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/metaladder-ascending-mathematical-solution","slug":"metaladder-ascending-mathematical-solution","title":"MetaLadder: Ascending Mathematical Solution Quality via Analogical-Problem Reasoning Transfer","date":"2025-03-19","arxiv_id":"2503.14891","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-consistency-for-llm-reasoning","slug":"temporal-consistency-for-llm-reasoning","title":"Temporal Consistency for LLM Reasoning Process Error Identification","date":"2025-03-18","arxiv_id":"2503.14495","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-llm-reasoning-with-iterative-dpo-a","slug":"enhancing-llm-reasoning-with-iterative-dpo-a","title":"Enhancing LLM Reasoning with Iterative DPO: A Comprehensive Empirical Investigation","date":"2025-03-17","arxiv_id":"2503.12854","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-reasoning-in-transformers-is","slug":"implicit-reasoning-in-transformers-is","title":"Implicit Reasoning in Transformers is Reasoning through Shortcuts","date":"2025-03-10","arxiv_id":"2503.07604","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/implicit-reasoning-in-transformers-is#ran","syntology_url":"https://syntology.ai/paper/2503.07604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07604"}},"official":{"repos":["TianheL/LM-Implicit-Reasoning"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vlrmbench-a-comprehensive-and-challenging","slug":"vlrmbench-a-comprehensive-and-challenging","title":"VLRMBench: A Comprehensive and Challenging Benchmark for Vision-Language Reward Models","date":"2025-03-10","arxiv_id":"2503.07478","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vlrmbench-a-comprehensive-and-challenging#ran","syntology_url":"https://syntology.ai/paper/2503.07478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07478"}},"official":{"repos":["jcruan519/vlrmbench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-atomic-step-decomposition-enhance-the","slug":"can-atomic-step-decomposition-enhance-the","title":"Can Atomic Step Decomposition Enhance the Self-structured Reasoning of Multimodal Large Models?","date":"2025-03-08","arxiv_id":"2503.06252","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/can-atomic-step-decomposition-enhance-the#ran","syntology_url":"https://syntology.ai/paper/2503.06252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06252"}},"official":{"repos":["quinn777/atomthink"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/routereval-a-comprehensive-benchmark-for","slug":"routereval-a-comprehensive-benchmark-for","title":"RouterEval: A Comprehensive Benchmark for Routing LLMs to Explore Model-level Scaling Up in LLMs","date":"2025-03-08","arxiv_id":"2503.10657","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/routereval-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2503.10657","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10657"}},"official":{"repos":["milkthink-lab/routereval"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/process-based-self-rewarding-language-models","slug":"process-based-self-rewarding-language-models","title":"Process-based Self-Rewarding Language Models","date":"2025-03-05","arxiv_id":"2503.03746","repositories_listed":1,"syntology":null},{"url":"/paper/an-efficient-and-precise-training-data","slug":"an-efficient-and-precise-training-data","title":"An Efficient and Precise Training Data Construction Framework for Process-supervised Reward Model in Mathematical Reasoning","date":"2025-03-04","arxiv_id":"2503.02382","repositories_listed":1,"syntology":null},{"url":"/paper/promptcot-synthesizing-olympiad-level","slug":"promptcot-synthesizing-olympiad-level","title":"PromptCoT: Synthesizing Olympiad-level Problems for Mathematical Reasoning in Large Language Models","date":"2025-03-04","arxiv_id":"2503.02324","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/promptcot-synthesizing-olympiad-level#ran","syntology_url":"https://syntology.ai/paper/2503.02324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.02324"}},"official":{"repos":["zhaoxlpku/promptcot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cot-uq-improving-response-wise-uncertainty","slug":"cot-uq-improving-response-wise-uncertainty","title":"CoT-UQ: Improving Response-wise Uncertainty Quantification in LLMs with Chain-of-Thought","date":"2025-02-24","arxiv_id":"2502.17214","repositories_listed":1,"syntology":null},{"url":"/paper/linguistic-generalizability-of-test-time","slug":"linguistic-generalizability-of-test-time","title":"Linguistic Generalizability of Test-Time Scaling in Mathematical Reasoning","date":"2025-02-24","arxiv_id":"2502.17407","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/linguistic-generalizability-of-test-time#ran","syntology_url":"https://syntology.ai/paper/2502.17407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17407"}},"official":{"repos":["gauss5930/mclm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/forgotten-polygons-multimodal-large-language","slug":"forgotten-polygons-multimodal-large-language","title":"Forgotten Polygons: Multimodal Large Language Models are Shape-Blind","date":"2025-02-21","arxiv_id":"2502.15969","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/forgotten-polygons-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2502.15969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15969"}},"official":{"repos":["rsinghlab/shape-blind"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/the-relationship-between-reasoning-and","slug":"the-relationship-between-reasoning-and","title":"The Relationship Between Reasoning and Performance in Large Language Models -- o3 (mini) Thinks Harder, Not Longer","date":"2025-02-21","arxiv_id":"2502.15631","repositories_listed":1,"syntology":null},{"url":"/paper/cer-confidence-enhanced-reasoning-in-llms","slug":"cer-confidence-enhanced-reasoning-in-llms","title":"CER: Confidence Enhanced Reasoning in LLMs","date":"2025-02-20","arxiv_id":"2502.14634","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":0,"n_instrument":9,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":15,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 9 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/cer-confidence-enhanced-reasoning-in-llms#ran","syntology_url":"https://syntology.ai/paper/2502.14634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14634"}},"official":{"repos":["sharif-ml-lab/CER"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptivestep-automatically-dividing-reasoning","slug":"adaptivestep-automatically-dividing-reasoning","title":"AdaptiveStep: Automatically Dividing Reasoning Step through Model Confidence","date":"2025-02-19","arxiv_id":"2502.13943","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptivestep-automatically-dividing-reasoning#ran","syntology_url":"https://syntology.ai/paper/2502.13943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13943"}},"official":{"repos":["lux0926/asprm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/proving-olympiad-inequalities-by-synergizing","slug":"proving-olympiad-inequalities-by-synergizing","title":"Proving Olympiad Inequalities by Synergizing LLMs and Symbolic Reasoning","date":"2025-02-19","arxiv_id":"2502.13834","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/proving-olympiad-inequalities-by-synergizing#ran","syntology_url":"https://syntology.ai/paper/2502.13834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13834"}},"official":{"repos":["lizn-zn/neqlips"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mathematical-reasoning-in-large-language","slug":"mathematical-reasoning-in-large-language","title":"Mathematical Reasoning in Large Language Models: Assessing Logical and Arithmetic Errors across Wide Numerical Ranges","date":"2025-02-12","arxiv_id":"2502.08680","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-fine-tuning-when-scaling-test-time","slug":"rethinking-fine-tuning-when-scaling-test-time","title":"Rethinking Fine-Tuning when Scaling Test-Time Compute: Limiting Confidence Improves Mathematical Reasoning","date":"2025-02-11","arxiv_id":"2502.07154","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rethinking-fine-tuning-when-scaling-test-time#ran","syntology_url":"https://syntology.ai/paper/2502.07154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07154"}},"official":{"repos":["allanraventos/refine"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-the-limit-of-outcome-reward-for","slug":"exploring-the-limit-of-outcome-reward-for","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning","date":"2025-02-10","arxiv_id":"2502.06781","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-the-limit-of-outcome-reward-for#ran","syntology_url":"https://syntology.ai/paper/2502.06781","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06781"}},"official":{"repos":["internlm/oreal"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kvtuner-sensitivity-aware-layer-wise-mixed","slug":"kvtuner-sensitivity-aware-layer-wise-mixed","title":"KVTuner: Sensitivity-Aware Layer-wise Mixed Precision KV Cache Quantization for Efficient and Nearly Lossless LLM Inference","date":"2025-02-06","arxiv_id":"2502.04420","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-for-multi-robot-systems","slug":"large-language-models-for-multi-robot-systems","title":"Large Language Models for Multi-Robot Systems: A Survey","date":"2025-02-06","arxiv_id":"2502.03814","repositories_listed":1,"syntology":null},{"url":"/paper/scoreflow-mastering-llm-agent-workflows-via","slug":"scoreflow-mastering-llm-agent-workflows-via","title":"ScoreFlow: Mastering LLM Agent Workflows via Score-based Preference Optimization","date":"2025-02-06","arxiv_id":"2502.04306","repositories_listed":1,"syntology":null},{"url":"/paper/reusing-embeddings-reproducible-reward-model","slug":"reusing-embeddings-reproducible-reward-model","title":"Reusing Embeddings: Reproducible Reward Model Research in Large Language Model Alignment without GPUs","date":"2025-02-04","arxiv_id":"2502.04357","repositories_listed":1,"syntology":null},{"url":"/paper/a-probabilistic-inference-approach-to","slug":"a-probabilistic-inference-approach-to","title":"A Probabilistic Inference Approach to Inference-Time Scaling of LLMs using Particle-Based Monte Carlo Methods","date":"2025-02-03","arxiv_id":"2502.01618","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-probabilistic-inference-approach-to#ran","syntology_url":"https://syntology.ai/paper/2502.01618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.01618"}},"official":{"repos":["Red-Hat-AI-Innovation-Team/its_hub"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-the-reasoning-gap-small-llms-can","slug":"bridging-the-reasoning-gap-small-llms-can","title":"Bridging the Reasoning Gap: Small LLMs Can Plan with Generalised Strategies","date":"2025-01-31","arxiv_id":"2501.18817","repositories_listed":1,"syntology":null},{"url":"/paper/critique-fine-tuning-learning-to-critique-is","slug":"critique-fine-tuning-learning-to-critique-is","title":"Critique Fine-Tuning: Learning to Critique is More Effective than Learning to Imitate","date":"2025-01-29","arxiv_id":"2501.17703","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/critique-fine-tuning-learning-to-critique-is#ran","syntology_url":"https://syntology.ai/paper/2501.17703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.17703"}},"official":null}},{"url":"/paper/o1-pruner-length-harmonizing-fine-tuning-for","slug":"o1-pruner-length-harmonizing-fine-tuning-for","title":"O1-Pruner: Length-Harmonizing Fine-Tuning for O1-Like Reasoning Pruning","date":"2025-01-22","arxiv_id":"2501.12570","repositories_listed":1,"syntology":null},{"url":"/paper/internlm-xcomposer2-5-reward-a-simple-yet","slug":"internlm-xcomposer2-5-reward-a-simple-yet","title":"InternLM-XComposer2.5-Reward: A Simple Yet Effective Multi-Modal Reward Model","date":"2025-01-21","arxiv_id":"2501.12368","repositories_listed":1,"syntology":null},{"url":"/paper/control-llm-controlled-evolution-for","slug":"control-llm-controlled-evolution-for","title":"Control LLM: Controlled Evolution for Intelligence Retention in LLM","date":"2025-01-19","arxiv_id":"2501.10979","repositories_listed":1,"syntology":null},{"url":"/paper/open-eyes-then-reason-fine-grained-visual","slug":"open-eyes-then-reason-fine-grained-visual","title":"Open Eyes, Then Reason: Fine-grained Visual Mathematical Understanding in MLLMs","date":"2025-01-11","arxiv_id":"2501.06430","repositories_listed":1,"syntology":null},{"url":"/paper/voxeval-benchmarking-the-knowledge","slug":"voxeval-benchmarking-the-knowledge","title":"VoxEval: Benchmarking the Knowledge Understanding Capabilities of End-to-End Spoken Language Models","date":"2025-01-09","arxiv_id":"2501.04962","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voxeval-benchmarking-the-knowledge#ran","syntology_url":"https://syntology.ai/paper/2501.04962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04962"}},"official":{"repos":["dreamtheater123/voxeval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ursa-understanding-and-verifying-chain-of","slug":"ursa-understanding-and-verifying-chain-of","title":"URSA: Understanding and Verifying Chain-of-thought Reasoning in Multimodal Mathematics","date":"2025-01-08","arxiv_id":"2501.04686","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ursa-understanding-and-verifying-chain-of#ran","syntology_url":"https://syntology.ai/paper/2501.04686","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04686"}},"official":{"repos":["URSA-MATH/URSA-MATH"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm2-let-large-language-models-harness-system","slug":"llm2-let-large-language-models-harness-system","title":"LLM2: Let Large Language Models Harness System 2 Reasoning","date":"2024-12-29","arxiv_id":"2412.20372","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-for-mathematical-1","slug":"large-language-models-for-mathematical-1","title":"Large Language Models for Mathematical Analysis","date":"2024-12-28","arxiv_id":"2501.00059","repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-mathematical-reasoning-advancing","slug":"multilingual-mathematical-reasoning-advancing","title":"Multilingual Mathematical Reasoning: Advancing Open-Source LLMs in Hindi and English","date":"2024-12-24","arxiv_id":"2412.18415","repositories_listed":1,"syntology":null},{"url":"/paper/b-star-monitoring-and-balancing-exploration","slug":"b-star-monitoring-and-balancing-exploration","title":"B-STaR: Monitoring and Balancing Exploration and Exploitation in Self-Taught Reasoners","date":"2024-12-23","arxiv_id":"2412.17256","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-sampling-scaling-inference","slug":"multi-agent-sampling-scaling-inference","title":"Multi-Agent Sampling: Scaling Inference Compute for Data Synthesis with Tree Search-Based Agentic Collaboration","date":"2024-12-22","arxiv_id":"2412.17061","repositories_listed":1,"syntology":null}],"record_sha256":"f503dea84145a796a3007bb9a99a6a8ff06372162d1fb838954b0e005700d53d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}