{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/math/papers/2","list_of":"/task/math","task":"Math","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":16,"rows_per_page":100,"rows":[101,200],"of":1596,"counts":{"archive_papers_tagged":1596,"with_a_code_link":765,"where_syntology_ran_a_sample":349,"not_listed_spam_title":0,"listed":1596,"listed_where_code_ran":349,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/math","prev":"/task/math","next":"/task/math/papers/3","papers":[{"url":"/paper/tighter-uniform-bounds-for-black-scholes","slug":"tighter-uniform-bounds-for-black-scholes","title":"Tighter 'uniform bounds for Black-Scholes implied volatility' and the applications to root-finding","date":"2023-02-17","arxiv_id":"2302.08758","repositories_listed":2,"syntology":null},{"url":"/paper/mathematical-capabilities-of-chatgpt-1","slug":"mathematical-capabilities-of-chatgpt-1","title":"Mathematical Capabilities of ChatGPT","date":"2023-01-31","arxiv_id":"2301.13867","repositories_listed":2,"syntology":null},{"url":"/paper/can-an-ai-win-ghana-s-national-science-and","slug":"can-an-ai-win-ghana-s-national-science-and","title":"Can an AI Win Ghana's National Science and Maths Quiz? An AI Grand Challenge for Education","date":"2023-01-30","arxiv_id":"2301.13089","repositories_listed":2,"syntology":null},{"url":"/paper/specializing-smaller-language-models-towards","slug":"specializing-smaller-language-models-towards","title":"Specializing Smaller Language Models towards Multi-Step Reasoning","date":"2023-01-30","arxiv_id":"2301.12726","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/specializing-smaller-language-models-towards#ran","syntology_url":"https://syntology.ai/paper/2301.12726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12726"}},"official":{"repos":["FranxYao/FlanT5-CoT-Specialization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unigeo-unifying-geometry-logical-reasoning","slug":"unigeo-unifying-geometry-logical-reasoning","title":"UniGeo: Unifying Geometry Logical Reasoning via Reformulating Mathematical Expression","date":"2022-12-06","arxiv_id":"2212.02746","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/unigeo-unifying-geometry-logical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2212.02746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02746"}},"official":{"repos":["chen-judge/unigeo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/program-of-thoughts-prompting-disentangling","slug":"program-of-thoughts-prompting-disentangling","title":"Program of Thoughts Prompting: Disentangling Computation from Reasoning for Numerical Reasoning Tasks","date":"2022-11-22","arxiv_id":"2211.12588","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/program-of-thoughts-prompting-disentangling#ran","syntology_url":"https://syntology.ai/paper/2211.12588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12588"}},"official":{"repos":["wenhuchen/program-of-thoughts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-prompt-learning-via-policy-gradient","slug":"dynamic-prompt-learning-via-policy-gradient","title":"Dynamic Prompt Learning via Policy Gradient for Semi-structured Mathematical Reasoning","date":"2022-09-29","arxiv_id":"2209.14610","repositories_listed":2,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/dynamic-prompt-learning-via-policy-gradient#ran","syntology_url":"https://syntology.ai/paper/2209.14610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.14610"}},"official":null}},{"url":"/paper/logicsolver-towards-interpretable-math-word","slug":"logicsolver-towards-interpretable-math-word","title":"LogicSolver: Towards Interpretable Math Word Problem Solving with Logical Prompt-enhanced Learning","date":"2022-05-17","arxiv_id":"2205.08232","repositories_listed":2,"syntology":null},{"url":"/paper/unbiased-math-word-problems-benchmark-for-1","slug":"unbiased-math-word-problems-benchmark-for-1","title":"Unbiased Math Word Problems Benchmark for Mitigating Solving Bias","date":"2022-05-17","arxiv_id":"2205.08108","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unbiased-math-word-problems-benchmark-for-1#ran","syntology_url":"https://syntology.ai/paper/2205.08108","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.08108"}},"official":{"repos":["yangzhch6/unbiasedmwp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-note-on-the-option-price-and-mass-at-zero","slug":"a-note-on-the-option-price-and-mass-at-zero","title":"A note on the option price and 'Mass at zero in the uncorrelated SABR model and implied volatility asymptotics'","date":"2020-11-01","arxiv_id":"2011.00557","repositories_listed":2,"syntology":null},{"url":"/paper/muscle-synergy-and-coupling-for-hand","slug":"muscle-synergy-and-coupling-for-hand","title":"A Relation Spectrum Inheriting Taylor Series: Muscle Synergy and Coupling for Hand","date":"2020-04-25","arxiv_id":"2004.11910","repositories_listed":2,"syntology":null},{"url":"/paper/integer-quantization-for-deep-learning","slug":"integer-quantization-for-deep-learning","title":"Integer Quantization for Deep Learning Inference: Principles and Empirical Evaluation","date":"2020-04-20","arxiv_id":"2004.09602","repositories_listed":2,"syntology":null},{"url":"/paper/injecting-numerical-reasoning-skills-into","slug":"injecting-numerical-reasoning-skills-into","title":"Injecting Numerical Reasoning Skills into Language Models","date":"2020-04-09","arxiv_id":"2004.04487","repositories_listed":2,"syntology":null},{"url":"/paper/multi-scale-attention-with-dense-encoder-for","slug":"multi-scale-attention-with-dense-encoder-for","title":"Multi-Scale Attention with Dense Encoder for Handwritten Mathematical Expression Recognition","date":"2018-01-05","arxiv_id":"1801.03530","repositories_listed":2,"syntology":null},{"url":"/paper/neural-machine-translation-and-sequence-to","slug":"neural-machine-translation-and-sequence-to","title":"Neural Machine Translation and Sequence-to-sequence Models: A Tutorial","date":"2017-03-05","arxiv_id":"1703.01619","repositories_listed":2,"syntology":null},{"url":"/paper/personalized-exercise-recommendation-with","slug":"personalized-exercise-recommendation-with","title":"Personalized Exercise Recommendation with Semantically-Grounded Knowledge Tracing","date":"2025-07-15","arxiv_id":"2507.11060","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-or-memorization-unreliable-results","slug":"reasoning-or-memorization-unreliable-results","title":"Reasoning or Memorization? Unreliable Results of Reinforcement Learning Due to Data Contamination","date":"2025-07-14","arxiv_id":"2507.10532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reasoning-or-memorization-unreliable-results#ran","syntology_url":"https://syntology.ai/paper/2507.10532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.10532"}},"official":{"repos":["wumingqi/LLM-Math-Evaluation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-practical-two-stage-recipe-for-mathematical","slug":"a-practical-two-stage-recipe-for-mathematical","title":"A Practical Two-Stage Recipe for Mathematical LLMs: Maximizing Accuracy with SFT and Efficiency with Reinforcement Learning","date":"2025-07-11","arxiv_id":"2507.08267","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-practical-two-stage-recipe-for-mathematical#ran","syntology_url":"https://syntology.ai/paper/2507.08267","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.08267"}},"official":{"repos":["analokmaus/kaggle-aimo2-fast-math-r1"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-delta-learning-hypothesis-preference","slug":"the-delta-learning-hypothesis-preference","title":"The Delta Learning Hypothesis: Preference Tuning on Weak Data can Yield Strong Gains","date":"2025-07-08","arxiv_id":"2507.06187","repositories_listed":1,"syntology":null},{"url":"/paper/llmthinkbench-towards-basic-math-reasoning","slug":"llmthinkbench-towards-basic-math-reasoning","title":"LLMThinkBench: Towards Basic Math Reasoning and Overthinking in Large Language Models","date":"2025-07-05","arxiv_id":"2507.04023","repositories_listed":1,"syntology":null},{"url":"/paper/effects-of-structure-on-reasoning-in-instance-1","slug":"effects-of-structure-on-reasoning-in-instance-1","title":"Effects of structure on reasoning in instance-level Self-Discover","date":"2025-07-04","arxiv_id":"2507.03347","repositories_listed":1,"syntology":null},{"url":"/paper/evoagentx-an-automated-framework-for-evolving","slug":"evoagentx-an-automated-framework-for-evolving","title":"EvoAgentX: An Automated Framework for Evolving Agentic Workflows","date":"2025-07-04","arxiv_id":"2507.03616","repositories_listed":1,"syntology":{"n":20,"n_ran":19,"n_constructed":0,"n_ran_checked":19,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":19,"n_pointer_only":16,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evoagentx-an-automated-framework-for-evolving#ran","syntology_url":"https://syntology.ai/paper/2507.03616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.03616"}},"official":{"repos":["evoagentx/evoagentx"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/energy-based-transformers-are-scalable","slug":"energy-based-transformers-are-scalable","title":"Energy-Based Transformers are Scalable Learners and Thinkers","date":"2025-07-02","arxiv_id":"2507.02092","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/energy-based-transformers-are-scalable#ran","syntology_url":"https://syntology.ai/paper/2507.02092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.02092"}},"official":{"repos":["alexiglad/EBT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/spiral-self-play-on-zero-sum-games","slug":"spiral-self-play-on-zero-sum-games","title":"SPIRAL: Self-Play on Zero-Sum Games Incentivizes Reasoning via Multi-Agent Multi-Turn Reinforcement Learning","date":"2025-06-30","arxiv_id":"2506.24119","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spiral-self-play-on-zero-sum-games#ran","syntology_url":"https://syntology.ai/paper/2506.24119","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.24119"}},"official":{"repos":["spiral-rl/spiral"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aalc-large-language-model-efficient-reasoning","slug":"aalc-large-language-model-efficient-reasoning","title":"AALC: Large Language Model Efficient Reasoning via Adaptive Accuracy-Length Control","date":"2025-06-25","arxiv_id":"2506.20160","repositories_listed":1,"syntology":null},{"url":"/paper/octothinker-mid-training-incentivizes","slug":"octothinker-mid-training-incentivizes","title":"OctoThinker: Mid-training Incentivizes Reinforcement Learning Scaling","date":"2025-06-25","arxiv_id":"2506.20512","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":4,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/octothinker-mid-training-incentivizes#ran","syntology_url":"https://syntology.ai/paper/2506.20512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20512"}},"official":{"repos":["gair-nlp/octothinker"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/confucius3-math-a-lightweight-high","slug":"confucius3-math-a-lightweight-high","title":"Confucius3-Math: A Lightweight High-Performance Reasoning LLM for Chinese K-12 Mathematics Learning","date":"2025-06-23","arxiv_id":"2506.18330","repositories_listed":1,"syntology":null},{"url":"/paper/evolving-prompts-in-context-an-open-ended","slug":"evolving-prompts-in-context-an-open-ended","title":"Evolving Prompts In-Context: An Open-ended, Self-replicating Perspective","date":"2025-06-22","arxiv_id":"2506.17930","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evolving-prompts-in-context-an-open-ended#ran","syntology_url":"https://syntology.ai/paper/2506.17930","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.17930"}},"official":{"repos":["jianyu-cs/promptquine"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ojbench-a-competition-level-code-benchmark","slug":"ojbench-a-competition-level-code-benchmark","title":"OJBench: A Competition Level Code Benchmark For Large Language Models","date":"2025-06-19","arxiv_id":"2506.16395","repositories_listed":1,"syntology":null},{"url":"/paper/agentgroupchat-v2-divide-and-conquer-is-what","slug":"agentgroupchat-v2-divide-and-conquer-is-what","title":"AgentGroupChat-V2: Divide-and-Conquer Is What LLM-Based Multi-Agent System Need","date":"2025-06-18","arxiv_id":"2506.15451","repositories_listed":1,"syntology":null},{"url":"/paper/essential-web-v1-0-24t-tokens-of-organized","slug":"essential-web-v1-0-24t-tokens-of-organized","title":"Essential-Web v1.0: 24T tokens of organized web data","date":"2025-06-17","arxiv_id":"2506.14111","repositories_listed":1,"syntology":null},{"url":"/paper/xolver-multi-agent-reasoning-with-holistic","slug":"xolver-multi-agent-reasoning-with-holistic","title":"Xolver: Multi-Agent Reasoning with Holistic Experience Learning Just Like an Olympiad Team","date":"2025-06-17","arxiv_id":"2506.14234","repositories_listed":1,"syntology":null},{"url":"/paper/steering-llm-thinking-with-budget-guidance","slug":"steering-llm-thinking-with-budget-guidance","title":"Steering LLM Thinking with Budget Guidance","date":"2025-06-16","arxiv_id":"2506.13752","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":11,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/steering-llm-thinking-with-budget-guidance#ran","syntology_url":"https://syntology.ai/paper/2506.13752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13752"}},"official":{"repos":["umass-embodied-agi/budgetguidance"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/treerl-llm-reinforcement-learning-with-on","slug":"treerl-llm-reinforcement-learning-with-on","title":"TreeRL: LLM Reinforcement Learning with On-Policy Tree Search","date":"2025-06-13","arxiv_id":"2506.11902","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/treerl-llm-reinforcement-learning-with-on#ran","syntology_url":"https://syntology.ai/paper/2506.11902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.11902"}},"official":{"repos":["thudm/treerl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-a-continue-thinking-token-for","slug":"learning-a-continue-thinking-token-for","title":"Learning a Continue-Thinking Token for Enhanced Test-Time Scaling","date":"2025-06-12","arxiv_id":"2506.11274","repositories_listed":1,"syntology":null},{"url":"/paper/recut-balancing-reasoning-length-and-accuracy","slug":"recut-balancing-reasoning-length-and-accuracy","title":"ReCUT: Balancing Reasoning Length and Accuracy in LLMs via Stepwise Trails and Preference Optimization","date":"2025-06-12","arxiv_id":"2506.10822","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recut-balancing-reasoning-length-and-accuracy#ran","syntology_url":"https://syntology.ai/paper/2506.10822","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.10822"}},"official":{"repos":["neuir/recut"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/spurious-rewards-rethinking-training-signals","slug":"spurious-rewards-rethinking-training-signals","title":"Spurious Rewards: Rethinking Training Signals in RLVR","date":"2025-06-12","arxiv_id":"2506.10947","repositories_listed":1,"syntology":null},{"url":"/paper/repo-replay-enhanced-policy-optimization","slug":"repo-replay-enhanced-policy-optimization","title":"RePO: Replay-Enhanced Policy Optimization","date":"2025-06-11","arxiv_id":"2506.09340","repositories_listed":1,"syntology":null},{"url":"/paper/resa-transparent-reasoning-models-via-saes","slug":"resa-transparent-reasoning-models-via-saes","title":"Resa: Transparent Reasoning Models via SAEs","date":"2025-06-11","arxiv_id":"2506.09967","repositories_listed":1,"syntology":null},{"url":"/paper/vicrit-a-verifiable-reinforcement-learning","slug":"vicrit-a-verifiable-reinforcement-learning","title":"ViCrit: A Verifiable Reinforcement Learning Proxy Task for Visual Perception in VLMs","date":"2025-06-11","arxiv_id":"2506.10128","repositories_listed":1,"syntology":null},{"url":"/paper/vision-matters-simple-visual-perturbations","slug":"vision-matters-simple-visual-perturbations","title":"Vision Matters: Simple Visual Perturbations Can Boost Multimodal Math Reasoning","date":"2025-06-11","arxiv_id":"2506.09736","repositories_listed":1,"syntology":null},{"url":"/paper/abstentionbench-reasoning-llms-fail-on","slug":"abstentionbench-reasoning-llms-fail-on","title":"AbstentionBench: Reasoning LLMs Fail on Unanswerable Questions","date":"2025-06-10","arxiv_id":"2506.09038","repositories_listed":1,"syntology":null},{"url":"/paper/sws-self-aware-weakness-driven-problem","slug":"sws-self-aware-weakness-driven-problem","title":"SwS: Self-aware Weakness-driven Problem Synthesis in Reinforcement Learning for LLM Reasoning","date":"2025-06-10","arxiv_id":"2506.08989","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sws-self-aware-weakness-driven-problem#ran","syntology_url":"https://syntology.ai/paper/2506.08989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08989"}},"official":{"repos":["mastervito/sws"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/play-to-generalize-learning-to-reason-through","slug":"play-to-generalize-learning-to-reason-through","title":"Play to Generalize: Learning to Reason Through Game Play","date":"2025-06-09","arxiv_id":"2506.08011","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/play-to-generalize-learning-to-reason-through#ran","syntology_url":"https://syntology.ai/paper/2506.08011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08011"}},"official":{"repos":["yunfeixie233/vigal"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/wethink-toward-general-purpose-vision","slug":"wethink-toward-general-purpose-vision","title":"WeThink: Toward General-purpose Vision-Language Reasoning via Reinforcement Learning","date":"2025-06-09","arxiv_id":"2506.07905","repositories_listed":1,"syntology":null},{"url":"/paper/spectral-derivatives","slug":"spectral-derivatives","title":"Spectral Derivatives","date":"2025-06-06","arxiv_id":"2506.06210","repositories_listed":1,"syntology":null},{"url":"/paper/mathematical-reasoning-for-unmanned-aerial","slug":"mathematical-reasoning-for-unmanned-aerial","title":"Mathematical Reasoning for Unmanned Aerial Vehicles: A RAG-Based Approach for Complex Arithmetic Reasoning","date":"2025-06-05","arxiv_id":"2506.04998","repositories_listed":1,"syntology":null},{"url":"/paper/mint-cot-enabling-interleaved-visual-tokens","slug":"mint-cot-enabling-interleaved-visual-tokens","title":"MINT-CoT: Enabling Interleaved Visual Tokens in Mathematical Chain-of-Thought Reasoning","date":"2025-06-05","arxiv_id":"2506.05331","repositories_listed":1,"syntology":null},{"url":"/paper/generating-pedagogically-meaningful-visuals","slug":"generating-pedagogically-meaningful-visuals","title":"Generating Pedagogically Meaningful Visuals for Math Word Problems: A New Benchmark and Analysis of Text-to-Image Models","date":"2025-06-04","arxiv_id":"2506.03735","repositories_listed":1,"syntology":null},{"url":"/paper/guided-speculative-inference-for-efficient","slug":"guided-speculative-inference-for-efficient","title":"Guided Speculative Inference for Efficient Test-Time Alignment of LLMs","date":"2025-06-04","arxiv_id":"2506.04118","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guided-speculative-inference-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2506.04118","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.04118"}},"official":{"repos":["j-geuter/gsi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/openthoughts-data-recipes-for-reasoning","slug":"openthoughts-data-recipes-for-reasoning","title":"OpenThoughts: Data Recipes for Reasoning Models","date":"2025-06-04","arxiv_id":"2506.04178","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/openthoughts-data-recipes-for-reasoning#ran","syntology_url":"https://syntology.ai/paper/2506.04178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.04178"}},"official":{"repos":["open-thoughts/open-thoughts"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/invariance-makes-llm-unlearning-resilient","slug":"invariance-makes-llm-unlearning-resilient","title":"Invariance Makes LLM Unlearning Resilient Even to Unanticipated Downstream Fine-Tuning","date":"2025-06-02","arxiv_id":"2506.01339","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/invariance-makes-llm-unlearning-resilient#ran","syntology_url":"https://syntology.ai/paper/2506.01339","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01339"}},"official":{"repos":["optml-group/unlearn-ilu"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/storm-born-a-challenging-mathematical","slug":"storm-born-a-challenging-mathematical","title":"STORM-BORN: A Challenging Mathematical Derivations Dataset Curated via a Human-in-the-Loop Multi-Agent Framework","date":"2025-06-02","arxiv_id":"2506.01531","repositories_listed":1,"syntology":null},{"url":"/paper/the-surprising-effectiveness-of-negative","slug":"the-surprising-effectiveness-of-negative","title":"The Surprising Effectiveness of Negative Reinforcement in LLM Reasoning","date":"2025-06-02","arxiv_id":"2506.01347","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-surprising-effectiveness-of-negative#ran","syntology_url":"https://syntology.ai/paper/2506.01347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01347"}},"official":{"repos":["tianhongzxy/rlvr-decomposed"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-thought-efficient-reasoning-via","slug":"a-thought-efficient-reasoning-via","title":"A*-Thought: Efficient Reasoning via Bidirectional Compression for Low-Resource Settings","date":"2025-05-30","arxiv_id":"2505.24550","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-thought-efficient-reasoning-via#ran","syntology_url":"https://syntology.ai/paper/2505.24550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24550"}},"official":{"repos":["ai9stars/astar-thought"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/agent-x-evaluating-deep-multimodal-reasoning","slug":"agent-x-evaluating-deep-multimodal-reasoning","title":"Agent-X: Evaluating Deep Multimodal Reasoning in Vision-Centric Agentic Tasks","date":"2025-05-30","arxiv_id":"2505.24876","repositories_listed":1,"syntology":null},{"url":"/paper/areal-a-large-scale-asynchronous","slug":"areal-a-large-scale-asynchronous","title":"AReaL: A Large-Scale Asynchronous Reinforcement Learning System for Language Reasoning","date":"2025-05-30","arxiv_id":"2505.24298","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/areal-a-large-scale-asynchronous#ran","syntology_url":"https://syntology.ai/paper/2505.24298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24298"}},"official":{"repos":["inclusionai/areal"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-negative-signals-reinforcement","slug":"harnessing-negative-signals-reinforcement","title":"Harnessing Negative Signals: Reinforcement Distillation from Teacher Data for LLM Reasoning","date":"2025-05-30","arxiv_id":"2505.24850","repositories_listed":1,"syntology":null},{"url":"/paper/mixed-r1-unified-reward-perspective-for","slug":"mixed-r1-unified-reward-perspective-for","title":"Mixed-R1: Unified Reward Perspective For Reasoning Capability in Multimodal Large Language Models","date":"2025-05-30","arxiv_id":"2505.24164","repositories_listed":1,"syntology":null},{"url":"/paper/silvr-a-simple-language-based-video-reasoning","slug":"silvr-a-simple-language-based-video-reasoning","title":"SiLVR: A Simple Language-based Video Reasoning Framework","date":"2025-05-30","arxiv_id":"2505.24869","repositories_listed":1,"syntology":null},{"url":"/paper/discriminative-policy-optimization-for-token","slug":"discriminative-policy-optimization-for-token","title":"Discriminative Policy Optimization for Token-Level Reward Models","date":"2025-05-29","arxiv_id":"2505.23363","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/discriminative-policy-optimization-for-token#ran","syntology_url":"https://syntology.ai/paper/2505.23363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23363"}},"official":{"repos":["homzer/q-rm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-performance-for-code-generation-on-noisy","slug":"llm-performance-for-code-generation-on-noisy","title":"LLM Performance for Code Generation on Noisy Tasks","date":"2025-05-29","arxiv_id":"2505.23598","repositories_listed":1,"syntology":null},{"url":"/paper/matharena-evaluating-llms-on-uncontaminated","slug":"matharena-evaluating-llms-on-uncontaminated","title":"MathArena: Evaluating LLMs on Uncontaminated Math Competitions","date":"2025-05-29","arxiv_id":"2505.23281","repositories_listed":1,"syntology":{"n":19,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/matharena-evaluating-llms-on-uncontaminated#ran","syntology_url":"https://syntology.ai/paper/2505.23281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23281"}},"official":{"repos":["eth-sri/matharena"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/asymob-algebraic-symbolic-mathematical","slug":"asymob-algebraic-symbolic-mathematical","title":"ASyMOB: Algebraic Symbolic Mathematical Operations Benchmark","date":"2025-05-28","arxiv_id":"2505.23851","repositories_listed":1,"syntology":null},{"url":"/paper/decomposing-elements-of-problem-solving-what","slug":"decomposing-elements-of-problem-solving-what","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","date":"2025-05-28","arxiv_id":"2505.22756","repositories_listed":1,"syntology":null},{"url":"/paper/skywork-open-reasoner-1-technical-report","slug":"skywork-open-reasoner-1-technical-report","title":"Skywork Open Reasoner 1 Technical Report","date":"2025-05-28","arxiv_id":"2505.22312","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/skywork-open-reasoner-1-technical-report#ran","syntology_url":"https://syntology.ai/paper/2505.22312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22312"}},"official":{"repos":["skyworkai/skywork-or1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-action-model-with-open-world","slug":"vision-language-action-model-with-open-world","title":"ChatVLA-2: Vision-Language-Action Model with Open-World Embodied Reasoning from Pretrained Knowledge","date":"2025-05-28","arxiv_id":"2505.21906","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vision-language-action-model-with-open-world#ran","syntology_url":"https://syntology.ai/paper/2505.21906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21906"}},"official":null}},{"url":"/paper/r2r-efficiently-navigating-divergent","slug":"r2r-efficiently-navigating-divergent","title":"R2R: Efficiently Navigating Divergent Reasoning Paths with Small-Large Model Token Routing","date":"2025-05-27","arxiv_id":"2505.21600","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r2r-efficiently-navigating-divergent#ran","syntology_url":"https://syntology.ai/paper/2505.21600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21600"}},"official":{"repos":["thu-nics/r2r"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/real-prover-retrieval-augmented-lean-prover","slug":"real-prover-retrieval-augmented-lean-prover","title":"REAL-Prover: Retrieval Augmented Lean Prover for Mathematical Reasoning","date":"2025-05-27","arxiv_id":"2505.20613","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcing-general-reasoning-without","slug":"reinforcing-general-reasoning-without","title":"Reinforcing General Reasoning without Verifiers","date":"2025-05-27","arxiv_id":"2505.21493","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcing-general-reasoning-without#ran","syntology_url":"https://syntology.ai/paper/2505.21493","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21493"}},"official":{"repos":["sail-sg/verifree"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/error-typing-for-smarter-rewards-improving","slug":"error-typing-for-smarter-rewards-improving","title":"Error Typing for Smarter Rewards: Improving Process Reward Models with Error-Aware Hierarchical Supervision","date":"2025-05-26","arxiv_id":"2505.19706","repositories_listed":1,"syntology":null},{"url":"/paper/hard-negative-contrastive-learning-for-fine","slug":"hard-negative-contrastive-learning-for-fine","title":"Hard Negative Contrastive Learning for Fine-Grained Geometric Understanding in Large Multimodal Models","date":"2025-05-26","arxiv_id":"2505.20152","repositories_listed":1,"syntology":null},{"url":"/paper/inference-time-alignment-in-continuous-space","slug":"inference-time-alignment-in-continuous-space","title":"Inference-time Alignment in Continuous Space","date":"2025-05-26","arxiv_id":"2505.20081","repositories_listed":1,"syntology":null},{"url":"/paper/mas-zero-designing-multi-agent-systems-with","slug":"mas-zero-designing-multi-agent-systems-with","title":"MAS-Zero: Designing Multi-Agent Systems with Zero Supervision","date":"2025-05-26","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unifying-multimodal-large-language-model","slug":"unifying-multimodal-large-language-model","title":"Unifying Multimodal Large Language Model Capabilities and Modalities via Model Merging","date":"2025-05-26","arxiv_id":"2505.19892","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2505.19892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19892"}},"official":{"repos":["walkerworldpeace/mllmerging"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmath-a-multilingual-benchmark-for","slug":"mmath-a-multilingual-benchmark-for","title":"MMATH: A Multilingual Benchmark for Mathematical Reasoning","date":"2025-05-25","arxiv_id":"2505.19126","repositories_listed":1,"syntology":null},{"url":"/paper/enumerate-conjecture-prove-formally-solving","slug":"enumerate-conjecture-prove-formally-solving","title":"Enumerate-Conjecture-Prove: Formally Solving Answer-Construction Problems in Math Competitions","date":"2025-05-24","arxiv_id":"2505.18492","repositories_listed":1,"syntology":null},{"url":"/paper/how-is-llm-reasoning-distracted-by-irrelevant","slug":"how-is-llm-reasoning-distracted-by-irrelevant","title":"How Is LLM Reasoning Distracted by Irrelevant Context? An Analysis Using a Controlled Benchmark","date":"2025-05-24","arxiv_id":"2505.18761","repositories_listed":1,"syntology":null},{"url":"/paper/decoupled-visual-interpretation-and","slug":"decoupled-visual-interpretation-and","title":"Decoupled Visual Interpretation and Linguistic Reasoning for Math Problem Solving","date":"2025-05-23","arxiv_id":"2505.17609","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decoupled-visual-interpretation-and#ran","syntology_url":"https://syntology.ai/paper/2505.17609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17609"}},"official":{"repos":["guozix/dvlr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rader-reasoning-aware-dense-retrieval-models","slug":"rader-reasoning-aware-dense-retrieval-models","title":"RaDeR: Reasoning-aware Dense Retrieval Models","date":"2025-05-23","arxiv_id":"2505.18405","repositories_listed":1,"syntology":null},{"url":"/paper/towards-revealing-the-effectiveness-of-small","slug":"towards-revealing-the-effectiveness-of-small","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","date":"2025-05-23","arxiv_id":"2505.17988","repositories_listed":1,"syntology":null},{"url":"/paper/value-guided-search-for-efficient-chain-of","slug":"value-guided-search-for-efficient-chain-of","title":"Value-Guided Search for Efficient Chain-of-Thought Reasoning","date":"2025-05-23","arxiv_id":"2505.17373","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/value-guided-search-for-efficient-chain-of#ran","syntology_url":"https://syntology.ai/paper/2505.17373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17373"}},"official":{"repos":["kaiwenw/value-guided-search"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/conciserl-conciseness-guided-reinforcement","slug":"conciserl-conciseness-guided-reinforcement","title":"ConciseRL: Conciseness-Guided Reinforcement Learning for Efficient Reasoning Models","date":"2025-05-22","arxiv_id":"2505.17250","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conciserl-conciseness-guided-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2505.17250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17250"}},"official":{"repos":["razvandu/conciserl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/equivpruner-boosting-efficiency-and-quality-1","slug":"equivpruner-boosting-efficiency-and-quality-1","title":"EquivPruner: Boosting Efficiency and Quality in LLM-Based Search via Action Pruning","date":"2025-05-22","arxiv_id":"2505.16312","repositories_listed":1,"syntology":null},{"url":"/paper/saturn-sat-based-reinforcement-learning-to","slug":"saturn-sat-based-reinforcement-learning-to","title":"SATURN: SAT-based Reinforcement Learning to Unleash Language Model Reasoning","date":"2025-05-22","arxiv_id":"2505.16368","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/saturn-sat-based-reinforcement-learning-to#ran","syntology_url":"https://syntology.ai/paper/2505.16368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16368"}},"official":{"repos":["gtxygyzb/saturn-code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"url":"/paper/unlearning-isn-t-deletion-investigating","slug":"unlearning-isn-t-deletion-investigating","title":"Unlearning Isn't Deletion: Investigating Reversibility of Machine Unlearning in LLMs","date":"2025-05-22","arxiv_id":"2505.16831","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/unlearning-isn-t-deletion-investigating#ran","syntology_url":"https://syntology.ai/paper/2505.16831","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16831"}},"official":{"repos":["xiaoyuxu1/representational_analysis_tools"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"url":"/paper/webagent-r1-training-web-agents-via-end-to","slug":"webagent-r1-training-web-agents-via-end-to","title":"WebAgent-R1: Training Web Agents via End-to-End Multi-Turn Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.16421","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/webagent-r1-training-web-agents-via-end-to#ran","syntology_url":"https://syntology.ai/paper/2505.16421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16421"}},"official":{"repos":["weizhepei/webagent-r1"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/x-mas-towards-building-multi-agent-systems","slug":"x-mas-towards-building-multi-agent-systems","title":"X-MAS: Towards Building Multi-Agent Systems with Heterogeneous LLMs","date":"2025-05-22","arxiv_id":"2505.16997","repositories_listed":1,"syntology":null},{"url":"/paper/how-should-we-enhance-the-safety-of-large","slug":"how-should-we-enhance-the-safety-of-large","title":"How Should We Enhance the Safety of Large Reasoning Models: An Empirical Study","date":"2025-05-21","arxiv_id":"2505.15404","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":6,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-should-we-enhance-the-safety-of-large#ran","syntology_url":"https://syntology.ai/paper/2505.15404","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15404"}},"official":{"repos":["thu-coai/lrm-safety-study"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-design-matters-a-self-design-multi-agent","slug":"meta-design-matters-a-self-design-multi-agent","title":"Meta-Design Matters: A Self-Design Multi-Agent System","date":"2025-05-21","arxiv_id":"2505.14996","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/meta-design-matters-a-self-design-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2505.14996","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14996"}},"official":null}},{"url":"/paper/mirb-mathematical-information-retrieval","slug":"mirb-mathematical-information-retrieval","title":"MIRB: Mathematical Information Retrieval Benchmark","date":"2025-05-21","arxiv_id":"2505.15585","repositories_listed":1,"syntology":null},{"url":"/paper/modelingagent-bridging-llms-and-mathematical","slug":"modelingagent-bridging-llms-and-mathematical","title":"ModelingAgent: Bridging LLMs and Mathematical Modeling for Real-World Challenges","date":"2025-05-21","arxiv_id":"2505.15068","repositories_listed":1,"syntology":null},{"url":"/paper/rl-tango-reinforcing-generator-and-verifier","slug":"rl-tango-reinforcing-generator-and-verifier","title":"RL Tango: Reinforcing Generator and Verifier Together for Language Reasoning","date":"2025-05-21","arxiv_id":"2505.15034","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rl-tango-reinforcing-generator-and-verifier#ran","syntology_url":"https://syntology.ai/paper/2505.15034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15034"}},"official":{"repos":["kaiwenzha/rl-tango"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-unreasonable-effectiveness-of-entropy","slug":"the-unreasonable-effectiveness-of-entropy","title":"The Unreasonable Effectiveness of Entropy Minimization in LLM Reasoning","date":"2025-05-21","arxiv_id":"2505.15134","repositories_listed":1,"syntology":null},{"url":"/paper/training-step-level-reasoning-verifiers-with","slug":"training-step-level-reasoning-verifiers-with","title":"Training Step-Level Reasoning Verifiers with Formal Verification Tools","date":"2025-05-21","arxiv_id":"2505.15960","repositories_listed":1,"syntology":null},{"url":"/paper/general-reasoner-advancing-llm-reasoning","slug":"general-reasoner-advancing-llm-reasoning","title":"General-Reasoner: Advancing LLM Reasoning Across All Domains","date":"2025-05-20","arxiv_id":"2505.14652","repositories_listed":1,"syntology":null},{"url":"/paper/let-s-verify-math-questions-step-by-step","slug":"let-s-verify-math-questions-step-by-step","title":"Let's Verify Math Questions Step by Step","date":"2025-05-20","arxiv_id":"2505.13903","repositories_listed":1,"syntology":null},{"url":"/paper/tinyv-reducing-false-negatives-in","slug":"tinyv-reducing-false-negatives-in","title":"TinyV: Reducing False Negatives in Verification Improves RL for LLM Reasoning","date":"2025-05-20","arxiv_id":"2505.14625","repositories_listed":1,"syntology":{"n":19,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/tinyv-reducing-false-negatives-in#ran","syntology_url":"https://syntology.ai/paper/2505.14625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14625"}},"official":{"repos":["uw-nsl/tinyv"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptthink-reasoning-models-can-learn-when-to","slug":"adaptthink-reasoning-models-can-learn-when-to","title":"AdaptThink: Reasoning Models Can Learn When to Think","date":"2025-05-19","arxiv_id":"2505.13417","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":8,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 2 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adaptthink-reasoning-models-can-learn-when-to#ran","syntology_url":"https://syntology.ai/paper/2505.13417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13417"}},"official":{"repos":["thu-keg/adaptthink"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/seek-in-the-dark-reasoning-via-test-time","slug":"seek-in-the-dark-reasoning-via-test-time","title":"Seek in the Dark: Reasoning via Test-Time Instance-Level Policy Gradient in Latent Space","date":"2025-05-19","arxiv_id":"2505.13308","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/seek-in-the-dark-reasoning-via-test-time#ran","syntology_url":"https://syntology.ai/paper/2505.13308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13308"}},"official":{"repos":["bigai-nlco/latentseek"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}}],"record_sha256":"e538ca48b85018929366593a63e63d239897de56105d20d0bc1e25113aa60d0a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}