{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mathematical-reasoning/papers/ran/1","list_of":"/task/mathematical-reasoning","task":"Mathematical Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":197,"counts":{"archive_papers_tagged":805,"with_a_code_link":395,"where_syntology_ran_a_sample":197,"not_listed_spam_title":0,"listed":805,"listed_where_code_ran":197,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":159,"every_run_a_failure_of_syntologys_instrument":38,"listed_with_a_run_with_no_instrument_failure":159,"listed_every_run_a_failure_of_syntologys_instrument":38,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mathematical-reasoning/papers/ran/1","prev":null,"next":"/task/mathematical-reasoning/papers/ran/2","papers":[{"url":"/paper/reasoning-or-memorization-unreliable-results","slug":"reasoning-or-memorization-unreliable-results","title":"Reasoning or Memorization? Unreliable Results of Reinforcement Learning Due to Data Contamination","date":"2025-07-14","arxiv_id":"2507.10532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reasoning-or-memorization-unreliable-results#ran","syntology_url":"https://syntology.ai/paper/2507.10532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.10532"}},"official":{"repos":["wumingqi/LLM-Math-Evaluation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-practical-two-stage-recipe-for-mathematical","slug":"a-practical-two-stage-recipe-for-mathematical","title":"A Practical Two-Stage Recipe for Mathematical LLMs: Maximizing Accuracy with SFT and Efficiency with Reinforcement Learning","date":"2025-07-11","arxiv_id":"2507.08267","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-practical-two-stage-recipe-for-mathematical#ran","syntology_url":"https://syntology.ai/paper/2507.08267","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.08267"}},"official":{"repos":["analokmaus/kaggle-aimo2-fast-math-r1"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/agentic-r1-distilled-dual-strategy-reasoning","slug":"agentic-r1-distilled-dual-strategy-reasoning","title":"Agentic-R1: Distilled Dual-Strategy Reasoning","date":"2025-07-08","arxiv_id":"2507.05707","repositories_listed":0,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/agentic-r1-distilled-dual-strategy-reasoning#ran","syntology_url":"https://syntology.ai/paper/2507.05707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.05707"}},"official":null}},{"url":"/paper/skywork-r1v3-technical-report","slug":"skywork-r1v3-technical-report","title":"Skywork-R1V3 Technical Report","date":"2025-07-08","arxiv_id":"2507.06167","repositories_listed":2,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/skywork-r1v3-technical-report#ran","syntology_url":"https://syntology.ai/paper/2507.06167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.06167"}},"official":{"repos":["SkyworkAI/Skywork-R1V"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/vrest-enhancing-reasoning-in-large-vision","slug":"vrest-enhancing-reasoning-in-large-vision","title":"VReST: Enhancing Reasoning in Large Vision-Language Models through Tree Search and Self-Reward Mechanism","date":"2025-06-10","arxiv_id":"2506.08691","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":4,"n_instrument":8,"n_unverified":3,"n_honours":3,"n_violates":1,"n_no_contract":0,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 1 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vrest-enhancing-reasoning-in-large-vision#ran","syntology_url":"https://syntology.ai/paper/2506.08691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08691"}},"official":{"repos":["garyjiajia/vrest"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/the-surprising-effectiveness-of-negative","slug":"the-surprising-effectiveness-of-negative","title":"The Surprising Effectiveness of Negative Reinforcement in LLM Reasoning","date":"2025-06-02","arxiv_id":"2506.01347","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-surprising-effectiveness-of-negative#ran","syntology_url":"https://syntology.ai/paper/2506.01347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01347"}},"official":{"repos":["tianhongzxy/rlvr-decomposed"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-hallucination-dilemma-factuality-aware","slug":"the-hallucination-dilemma-factuality-aware","title":"The Hallucination Dilemma: Factuality-Aware Reinforcement Learning for Large Reasoning Models","date":"2025-05-30","arxiv_id":"2505.24630","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-hallucination-dilemma-factuality-aware#ran","syntology_url":"https://syntology.ai/paper/2505.24630","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24630"}},"official":{"repos":["nusnlp/fspo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/matharena-evaluating-llms-on-uncontaminated","slug":"matharena-evaluating-llms-on-uncontaminated","title":"MathArena: Evaluating LLMs on Uncontaminated Math Competitions","date":"2025-05-29","arxiv_id":"2505.23281","repositories_listed":1,"syntology":{"n":19,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/matharena-evaluating-llms-on-uncontaminated#ran","syntology_url":"https://syntology.ai/paper/2505.23281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23281"}},"official":{"repos":["eth-sri/matharena"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/discriminative-policy-optimization-for-token","slug":"discriminative-policy-optimization-for-token","title":"Discriminative Policy Optimization for Token-Level Reward Models","date":"2025-05-29","arxiv_id":"2505.23363","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/discriminative-policy-optimization-for-token#ran","syntology_url":"https://syntology.ai/paper/2505.23363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23363"}},"official":{"repos":["homzer/q-rm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deeptheorem-advancing-llm-reasoning-for","slug":"deeptheorem-advancing-llm-reasoning-for","title":"DeepTheorem: Advancing LLM Reasoning for Theorem Proving Through Natural Language and Reinforcement Learning","date":"2025-05-29","arxiv_id":"2505.23754","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deeptheorem-advancing-llm-reasoning-for#ran","syntology_url":"https://syntology.ai/paper/2505.23754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23754"}},"official":{"repos":["jiahao004/deeptheorem"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-action-model-with-open-world","slug":"vision-language-action-model-with-open-world","title":"ChatVLA-2: Vision-Language-Action Model with Open-World Embodied Reasoning from Pretrained Knowledge","date":"2025-05-28","arxiv_id":"2505.21906","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vision-language-action-model-with-open-world#ran","syntology_url":"https://syntology.ai/paper/2505.21906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21906"}},"official":null}},{"url":"/paper/reinforcing-general-reasoning-without","slug":"reinforcing-general-reasoning-without","title":"Reinforcing General Reasoning without Verifiers","date":"2025-05-27","arxiv_id":"2505.21493","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcing-general-reasoning-without#ran","syntology_url":"https://syntology.ai/paper/2505.21493","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21493"}},"official":{"repos":["sail-sg/verifree"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/limopro-reasoning-refinement-for-efficient","slug":"limopro-reasoning-refinement-for-efficient","title":"LIMOPro: Reasoning Refinement for Efficient and Effective Test-time Scaling","date":"2025-05-25","arxiv_id":"2505.19187","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/limopro-reasoning-refinement-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2505.19187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19187"}},"official":{"repos":["gair-nlp/limopro"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tropical-attention-neural-algorithmic","slug":"tropical-attention-neural-algorithmic","title":"Tropical Attention: Neural Algorithmic Reasoning for Combinatorial Algorithms","date":"2025-05-22","arxiv_id":"2505.17190","repositories_listed":0,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tropical-attention-neural-algorithmic#ran","syntology_url":"https://syntology.ai/paper/2505.17190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17190"}},"official":null}},{"url":"/paper/rl-tango-reinforcing-generator-and-verifier","slug":"rl-tango-reinforcing-generator-and-verifier","title":"RL Tango: Reinforcing Generator and Verifier Together for Language Reasoning","date":"2025-05-21","arxiv_id":"2505.15034","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rl-tango-reinforcing-generator-and-verifier#ran","syntology_url":"https://syntology.ai/paper/2505.15034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15034"}},"official":{"repos":["kaiwenzha/rl-tango"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/trajectory-bellman-residual-minimization-a","slug":"trajectory-bellman-residual-minimization-a","title":"Trajectory Bellman Residual Minimization: A Simple Value-Based Method for LLM Reasoning","date":"2025-05-21","arxiv_id":"2505.15311","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trajectory-bellman-residual-minimization-a#ran","syntology_url":"https://syntology.ai/paper/2505.15311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15311"}},"official":null}},{"url":"/paper/deepeyes-incentivizing-thinking-with-images","slug":"deepeyes-incentivizing-thinking-with-images","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14362","repositories_listed":1,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deepeyes-incentivizing-thinking-with-images#ran","syntology_url":"https://syntology.ai/paper/2505.14362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14362"}},"official":{"repos":["visual-agent/deepeyes"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-prm-enhancing-multimodal-mathematical","slug":"mm-prm-enhancing-multimodal-mathematical","title":"MM-PRM: Enhancing Multimodal Mathematical Reasoning with Scalable Step-Level Supervision","date":"2025-05-19","arxiv_id":"2505.13427","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mm-prm-enhancing-multimodal-mathematical#ran","syntology_url":"https://syntology.ai/paper/2505.13427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13427"}},"official":{"repos":["modalminds/mm-prm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/disco-reinforcing-large-reasoning-models-with","slug":"disco-reinforcing-large-reasoning-models-with","title":"DisCO: Reinforcing Large Reasoning Models with Discriminative Constrained Optimization","date":"2025-05-18","arxiv_id":"2505.12366","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/disco-reinforcing-large-reasoning-models-with#ran","syntology_url":"https://syntology.ai/paper/2505.12366","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12366"}},"official":{"repos":["optimization-ai/disco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/marge-improving-math-reasoning-for-llms-with","slug":"marge-improving-math-reasoning-for-llms-with","title":"MARGE: Improving Math Reasoning for LLMs with Guided Exploration","date":"2025-05-18","arxiv_id":"2505.12500","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/marge-improving-math-reasoning-for-llms-with#ran","syntology_url":"https://syntology.ai/paper/2505.12500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12500"}},"official":{"repos":["georgao35/marge"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","unlocated"]}}},{"url":"/paper/token-level-uncertainty-estimation-for-large","slug":"token-level-uncertainty-estimation-for-large","title":"Token-Level Uncertainty Estimation for Large Language Model Reasoning","date":"2025-05-16","arxiv_id":"2505.11737","repositories_listed":0,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/token-level-uncertainty-estimation-for-large#ran","syntology_url":"https://syntology.ai/paper/2505.11737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11737"}},"official":null}},{"url":"/paper/mathcoder-vl-bridging-vision-and-code-for","slug":"mathcoder-vl-bridging-vision-and-code-for","title":"MathCoder-VL: Bridging Vision and Code for Enhanced Multimodal Mathematical Reasoning","date":"2025-05-15","arxiv_id":"2505.10557","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathcoder-vl-bridging-vision-and-code-for#ran","syntology_url":"https://syntology.ai/paper/2505.10557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10557"}},"official":{"repos":["mathllm/mathcoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dra-grpo-exploring-diversity-aware-reward","slug":"dra-grpo-exploring-diversity-aware-reward","title":"DRA-GRPO: Exploring Diversity-Aware Reward Adjustment for R1-Zero-Like Training of Large Language Models","date":"2025-05-14","arxiv_id":"2505.09655","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dra-grpo-exploring-diversity-aware-reward#ran","syntology_url":"https://syntology.ai/paper/2505.09655","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.09655"}},"official":{"repos":["xiwenc1/dra-grpo"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/agent-rl-scaling-law-agent-rl-with","slug":"agent-rl-scaling-law-agent-rl-with","title":"Agent RL Scaling Law: Agent RL with Spontaneous Code Execution for Mathematical Problem Solving","date":"2025-05-12","arxiv_id":"2505.07773","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agent-rl-scaling-law-agent-rl-with#ran","syntology_url":"https://syntology.ai/paper/2505.07773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07773"}},"official":{"repos":["anonymize-author/agentrl","yyht/openrlhf_async_pipline"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crosslingual-reasoning-through-test-time","slug":"crosslingual-reasoning-through-test-time","title":"Crosslingual Reasoning through Test-Time Scaling","date":"2025-05-08","arxiv_id":"2505.05408","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/crosslingual-reasoning-through-test-time#ran","syntology_url":"https://syntology.ai/paper/2505.05408","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05408"}},"official":{"repos":["BatsResearch/crosslingual-test-time-scaling"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/absolute-zero-reinforced-self-play-reasoning","slug":"absolute-zero-reinforced-self-play-reasoning","title":"Absolute Zero: Reinforced Self-play Reasoning with Zero Data","date":"2025-05-06","arxiv_id":"2505.03335","repositories_listed":3,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":6,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/absolute-zero-reinforced-self-play-reasoning#ran","syntology_url":"https://syntology.ai/paper/2505.03335","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.03335"}},"official":{"repos":["LeapLabTHU/Absolute-Zero-Reasoner","volcengine/verl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/formalmath-benchmarking-formal-mathematical","slug":"formalmath-benchmarking-formal-mathematical","title":"FormalMATH: Benchmarking Formal Mathematical Reasoning of Large Language Models","date":"2025-05-05","arxiv_id":"2505.02735","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/formalmath-benchmarking-formal-mathematical#ran","syntology_url":"https://syntology.ai/paper/2505.02735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02735"}},"official":{"repos":["sphere-ai-lab/formalmath-bench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rewriting-pre-training-data-boosts-llm","slug":"rewriting-pre-training-data-boosts-llm","title":"Rewriting Pre-Training Data Boosts LLM Performance in Math and Code","date":"2025-05-05","arxiv_id":"2505.02881","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rewriting-pre-training-data-boosts-llm#ran","syntology_url":"https://syntology.ai/paper/2505.02881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02881"}},"official":{"repos":["rioyokotalab/swallow-code-math"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-reasoning-in-large","slug":"reinforcement-learning-for-reasoning-in-large","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","date":"2025-04-29","arxiv_id":"2504.20571","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":9,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/reinforcement-learning-for-reasoning-in-large#ran","syntology_url":"https://syntology.ai/paper/2504.20571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20571"}},"official":{"repos":["ypwang61/one-shot-rlvr"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-evaluative-thinking-meta-policy","slug":"toward-evaluative-thinking-meta-policy","title":"Toward Evaluative Thinking: Meta Policy Optimization with Evolving Reward Models","date":"2025-04-28","arxiv_id":"2504.20157","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":15,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/toward-evaluative-thinking-meta-policy#ran","syntology_url":"https://syntology.ai/paper/2504.20157","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20157"}},"official":{"repos":["minnesotanlp/mpo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/polymath-evaluating-mathematical-reasoning-in","slug":"polymath-evaluating-mathematical-reasoning-in","title":"PolyMath: Evaluating Mathematical Reasoning in Multilingual Contexts","date":"2025-04-25","arxiv_id":"2504.18428","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/polymath-evaluating-mathematical-reasoning-in#ran","syntology_url":"https://syntology.ai/paper/2504.18428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.18428"}},"official":null}},{"url":"/paper/aimo-2-winning-solution-building-state-of-the","slug":"aimo-2-winning-solution-building-state-of-the","title":"AIMO-2 Winning Solution: Building State-of-the-Art Mathematical Reasoning Models with OpenMathReasoning dataset","date":"2025-04-23","arxiv_id":"2504.16891","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/aimo-2-winning-solution-building-state-of-the#ran","syntology_url":"https://syntology.ai/paper/2504.16891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16891"}},"official":null}},{"url":"/paper/deepmath-103k-a-large-scale-challenging","slug":"deepmath-103k-a-large-scale-challenging","title":"DeepMath-103K: A Large-Scale, Challenging, Decontaminated, and Verifiable Mathematical Dataset for Advancing Reasoning","date":"2025-04-15","arxiv_id":"2504.11456","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepmath-103k-a-large-scale-challenging#ran","syntology_url":"https://syntology.ai/paper/2504.11456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11456"}},"official":{"repos":["zwhe99/deepmath"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/two-heads-are-better-than-one-test-time-1","slug":"two-heads-are-better-than-one-test-time-1","title":"Two Heads are Better Than One: Test-time Scaling of Multi-agent Collaborative Reasoning","date":"2025-04-14","arxiv_id":"2504.09772","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/two-heads-are-better-than-one-test-time-1#ran","syntology_url":"https://syntology.ai/paper/2504.09772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.09772"}},"official":{"repos":["jincan333/MAS-TTS"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grpo-lead-a-difficulty-aware-reinforcement-1","slug":"grpo-lead-a-difficulty-aware-reinforcement-1","title":"GRPO-LEAD: A Difficulty-Aware Reinforcement Learning Approach for Concise Mathematical Reasoning in Language Models","date":"2025-04-13","arxiv_id":"2504.09696","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/grpo-lead-a-difficulty-aware-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2504.09696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.09696"}},"official":{"repos":["aeroplanepaper/GRPO-LEAD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/echo-chamber-rl-post-training-amplifies","slug":"echo-chamber-rl-post-training-amplifies","title":"Echo Chamber: RL Post-training Amplifies Behaviors Learned in Pretraining","date":"2025-04-10","arxiv_id":"2504.07912","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/echo-chamber-rl-post-training-amplifies#ran","syntology_url":"https://syntology.ai/paper/2504.07912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07912"}},"official":{"repos":["rosieyzh/openrlhf-pretrain"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/right-question-is-already-half-the-answer","slug":"right-question-is-already-half-the-answer","title":"Right Question is Already Half the Answer: Fully Unsupervised LLM Reasoning Incentivization","date":"2025-04-08","arxiv_id":"2504.05812","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/right-question-is-already-half-the-answer#ran","syntology_url":"https://syntology.ai/paper/2504.05812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05812"}},"official":{"repos":["qingyangzhang/empo"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/efficient-reinforcement-finetuning-via","slug":"efficient-reinforcement-finetuning-via","title":"Efficient Reinforcement Finetuning via Adaptive Curriculum Learning","date":"2025-04-07","arxiv_id":"2504.05520","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/efficient-reinforcement-finetuning-via#ran","syntology_url":"https://syntology.ai/paper/2504.05520","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05520"}},"official":{"repos":["uscnlp-lime/verl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/megamath-pushing-the-limits-of-open-math","slug":"megamath-pushing-the-limits-of-open-math","title":"MegaMath: Pushing the Limits of Open Math Corpora","date":"2025-04-03","arxiv_id":"2504.02807","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/megamath-pushing-the-limits-of-open-math#ran","syntology_url":"https://syntology.ai/paper/2504.02807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02807"}},"official":{"repos":["llm360/megamath"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/r-prm-reasoning-driven-process-reward","slug":"r-prm-reasoning-driven-process-reward","title":"R-PRM: Reasoning-Driven Process Reward Modeling","date":"2025-03-27","arxiv_id":"2503.21295","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r-prm-reasoning-driven-process-reward#ran","syntology_url":"https://syntology.ai/paper/2503.21295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21295"}},"official":{"repos":["njunlp/r-prm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/challenging-the-boundaries-of-reasoning-an","slug":"challenging-the-boundaries-of-reasoning-an","title":"Challenging the Boundaries of Reasoning: An Olympiad-Level Math Benchmark for Large Language Models","date":"2025-03-27","arxiv_id":"2503.21380","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/challenging-the-boundaries-of-reasoning-an#ran","syntology_url":"https://syntology.ai/paper/2503.21380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21380"}},"official":{"repos":["RUCAIBox/Slow_Thinking_with_LLMs","rucaibox/olymmath"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/accelerate-parallelizable-reasoning-via","slug":"accelerate-parallelizable-reasoning-via","title":"Accelerate Parallelizable Reasoning via Parallel Decoding within One Sequence","date":"2025-03-26","arxiv_id":"2503.20533","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":5,"n_ran_checked":5,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"9 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/accelerate-parallelizable-reasoning-via#ran","syntology_url":"https://syntology.ai/paper/2503.20533","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20533"}},"official":{"repos":["yuyijiong/parallel-decoding-in-one-sequence"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/trajectory-balance-with-asynchrony-decoupling","slug":"trajectory-balance-with-asynchrony-decoupling","title":"Trajectory Balance with Asynchrony: Decoupling Exploration and Learning for Fast, Scalable LLM Post-Training","date":"2025-03-24","arxiv_id":"2503.18929","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/trajectory-balance-with-asynchrony-decoupling#ran","syntology_url":"https://syntology.ai/paper/2503.18929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18929"}},"official":null}},{"url":"/paper/mathfusion-enhancing-mathematic-problem","slug":"mathfusion-enhancing-mathematic-problem","title":"MathFusion: Enhancing Mathematic Problem-solving of LLM through Instruction Fusion","date":"2025-03-20","arxiv_id":"2503.16212","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mathfusion-enhancing-mathematic-problem#ran","syntology_url":"https://syntology.ai/paper/2503.16212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16212"}},"official":{"repos":["qizhipei/mathfusion"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-reasoning-in-small","slug":"reinforcement-learning-for-reasoning-in-small","title":"Reinforcement Learning for Reasoning in Small LLMs: What Works and What Doesn't","date":"2025-03-20","arxiv_id":"2503.16219","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-for-reasoning-in-small#ran","syntology_url":"https://syntology.ai/paper/2503.16219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16219"}},"official":{"repos":["knoveleng/open-rs"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vlrmbench-a-comprehensive-and-challenging","slug":"vlrmbench-a-comprehensive-and-challenging","title":"VLRMBench: A Comprehensive and Challenging Benchmark for Vision-Language Reward Models","date":"2025-03-10","arxiv_id":"2503.07478","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vlrmbench-a-comprehensive-and-challenging#ran","syntology_url":"https://syntology.ai/paper/2503.07478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07478"}},"official":{"repos":["jcruan519/vlrmbench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/implicit-reasoning-in-transformers-is","slug":"implicit-reasoning-in-transformers-is","title":"Implicit Reasoning in Transformers is Reasoning through Shortcuts","date":"2025-03-10","arxiv_id":"2503.07604","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/implicit-reasoning-in-transformers-is#ran","syntology_url":"https://syntology.ai/paper/2503.07604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07604"}},"official":{"repos":["TianheL/LM-Implicit-Reasoning"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-atomic-step-decomposition-enhance-the","slug":"can-atomic-step-decomposition-enhance-the","title":"Can Atomic Step Decomposition Enhance the Self-structured Reasoning of Multimodal Large Models?","date":"2025-03-08","arxiv_id":"2503.06252","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/can-atomic-step-decomposition-enhance-the#ran","syntology_url":"https://syntology.ai/paper/2503.06252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06252"}},"official":{"repos":["quinn777/atomthink"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/routereval-a-comprehensive-benchmark-for","slug":"routereval-a-comprehensive-benchmark-for","title":"RouterEval: A Comprehensive Benchmark for Routing LLMs to Explore Model-level Scaling Up in LLMs","date":"2025-03-08","arxiv_id":"2503.10657","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/routereval-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2503.10657","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10657"}},"official":{"repos":["milkthink-lab/routereval"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pi-gps-enhancing-geometry-problem-solving-by","slug":"pi-gps-enhancing-geometry-problem-solving-by","title":"Pi-GPS: Enhancing Geometry Problem Solving by Unleashing the Power of Diagrammatic Information","date":"2025-03-07","arxiv_id":"2503.05543","repositories_listed":0,"syntology":{"n":7,"n_ran":7,"n_constructed":1,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":3,"n_no_contract":1,"n_pointer_only":7,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 3 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pi-gps-enhancing-geometry-problem-solving-by#ran","syntology_url":"https://syntology.ai/paper/2503.05543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05543"}},"official":null}},{"url":"/paper/promptcot-synthesizing-olympiad-level","slug":"promptcot-synthesizing-olympiad-level","title":"PromptCoT: Synthesizing Olympiad-level Problems for Mathematical Reasoning in Large Language Models","date":"2025-03-04","arxiv_id":"2503.02324","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/promptcot-synthesizing-olympiad-level#ran","syntology_url":"https://syntology.ai/paper/2503.02324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.02324"}},"official":{"repos":["zhaoxlpku/promptcot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-rewarding-correction-for-mathematical","slug":"self-rewarding-correction-for-mathematical","title":"Self-rewarding correction for mathematical reasoning","date":"2025-02-26","arxiv_id":"2502.19613","repositories_listed":2,"syntology":{"n":16,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/self-rewarding-correction-for-mathematical#ran","syntology_url":"https://syntology.ai/paper/2502.19613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19613"}},"official":{"repos":["rlhflow/self-rewarding-reasoning-llm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/linguistic-generalizability-of-test-time","slug":"linguistic-generalizability-of-test-time","title":"Linguistic Generalizability of Test-Time Scaling in Mathematical Reasoning","date":"2025-02-24","arxiv_id":"2502.17407","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/linguistic-generalizability-of-test-time#ran","syntology_url":"https://syntology.ai/paper/2502.17407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17407"}},"official":{"repos":["gauss5930/mclm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/forgotten-polygons-multimodal-large-language","slug":"forgotten-polygons-multimodal-large-language","title":"Forgotten Polygons: Multimodal Large Language Models are Shape-Blind","date":"2025-02-21","arxiv_id":"2502.15969","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/forgotten-polygons-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2502.15969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15969"}},"official":{"repos":["rsinghlab/shape-blind"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/cer-confidence-enhanced-reasoning-in-llms","slug":"cer-confidence-enhanced-reasoning-in-llms","title":"CER: Confidence Enhanced Reasoning in LLMs","date":"2025-02-20","arxiv_id":"2502.14634","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":0,"n_instrument":9,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":15,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 9 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/cer-confidence-enhanced-reasoning-in-llms#ran","syntology_url":"https://syntology.ai/paper/2502.14634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14634"}},"official":{"repos":["sharif-ml-lab/CER"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/proving-olympiad-inequalities-by-synergizing","slug":"proving-olympiad-inequalities-by-synergizing","title":"Proving Olympiad Inequalities by Synergizing LLMs and Symbolic Reasoning","date":"2025-02-19","arxiv_id":"2502.13834","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/proving-olympiad-inequalities-by-synergizing#ran","syntology_url":"https://syntology.ai/paper/2502.13834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13834"}},"official":{"repos":["lizn-zn/neqlips"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptivestep-automatically-dividing-reasoning","slug":"adaptivestep-automatically-dividing-reasoning","title":"AdaptiveStep: Automatically Dividing Reasoning Step through Model Confidence","date":"2025-02-19","arxiv_id":"2502.13943","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptivestep-automatically-dividing-reasoning#ran","syntology_url":"https://syntology.ai/paper/2502.13943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13943"}},"official":{"repos":["lux0926/asprm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-fine-tuning-when-scaling-test-time","slug":"rethinking-fine-tuning-when-scaling-test-time","title":"Rethinking Fine-Tuning when Scaling Test-Time Compute: Limiting Confidence Improves Mathematical Reasoning","date":"2025-02-11","arxiv_id":"2502.07154","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rethinking-fine-tuning-when-scaling-test-time#ran","syntology_url":"https://syntology.ai/paper/2502.07154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07154"}},"official":{"repos":["allanraventos/refine"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-the-limit-of-outcome-reward-for","slug":"exploring-the-limit-of-outcome-reward-for","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning","date":"2025-02-10","arxiv_id":"2502.06781","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-the-limit-of-outcome-reward-for#ran","syntology_url":"https://syntology.ai/paper/2502.06781","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06781"}},"official":{"repos":["internlm/oreal"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-probabilistic-inference-approach-to","slug":"a-probabilistic-inference-approach-to","title":"A Probabilistic Inference Approach to Inference-Time Scaling of LLMs using Particle-Based Monte Carlo Methods","date":"2025-02-03","arxiv_id":"2502.01618","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-probabilistic-inference-approach-to#ran","syntology_url":"https://syntology.ai/paper/2502.01618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.01618"}},"official":{"repos":["Red-Hat-AI-Innovation-Team/its_hub"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/s1-simple-test-time-scaling","slug":"s1-simple-test-time-scaling","title":"s1: Simple test-time scaling","date":"2025-01-31","arxiv_id":"2501.19393","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/s1-simple-test-time-scaling#ran","syntology_url":"https://syntology.ai/paper/2501.19393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19393"}},"official":{"repos":["simplescaling/s1"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/critique-fine-tuning-learning-to-critique-is","slug":"critique-fine-tuning-learning-to-critique-is","title":"Critique Fine-Tuning: Learning to Critique is More Effective than Learning to Imitate","date":"2025-01-29","arxiv_id":"2501.17703","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/critique-fine-tuning-learning-to-critique-is#ran","syntology_url":"https://syntology.ai/paper/2501.17703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.17703"}},"official":null}},{"url":"/paper/voxeval-benchmarking-the-knowledge","slug":"voxeval-benchmarking-the-knowledge","title":"VoxEval: Benchmarking the Knowledge Understanding Capabilities of End-to-End Spoken Language Models","date":"2025-01-09","arxiv_id":"2501.04962","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voxeval-benchmarking-the-knowledge#ran","syntology_url":"https://syntology.ai/paper/2501.04962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04962"}},"official":{"repos":["dreamtheater123/voxeval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ursa-understanding-and-verifying-chain-of","slug":"ursa-understanding-and-verifying-chain-of","title":"URSA: Understanding and Verifying Chain-of-thought Reasoning in Multimodal Mathematics","date":"2025-01-08","arxiv_id":"2501.04686","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ursa-understanding-and-verifying-chain-of#ran","syntology_url":"https://syntology.ai/paper/2501.04686","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04686"}},"official":{"repos":["URSA-MATH/URSA-MATH"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-for-llm-multi","slug":"offline-reinforcement-learning-for-llm-multi","title":"Offline Reinforcement Learning for LLM Multi-Step Reasoning","date":"2024-12-20","arxiv_id":"2412.16145","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-reinforcement-learning-for-llm-multi#ran","syntology_url":"https://syntology.ai/paper/2412.16145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.16145"}},"official":{"repos":["jwhj/oreo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/qwen2-5-technical-report","slug":"qwen2-5-technical-report","title":"Qwen2.5 Technical Report","date":"2024-12-19","arxiv_id":"2412.15115","repositories_listed":6,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/qwen2-5-technical-report#ran","syntology_url":"https://syntology.ai/paper/2412.15115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15115"}},"official":{"repos":["qwenlm/qwen2.5"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/entropy-regularized-process-reward-model","slug":"entropy-regularized-process-reward-model","title":"Entropy-Regularized Process Reward Model","date":"2024-12-15","arxiv_id":"2412.11006","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/entropy-regularized-process-reward-model#ran","syntology_url":"https://syntology.ai/paper/2412.11006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11006"}},"official":{"repos":["hanningzhang/er-prm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/processbench-identifying-process-errors-in","slug":"processbench-identifying-process-errors-in","title":"ProcessBench: Identifying Process Errors in Mathematical Reasoning","date":"2024-12-09","arxiv_id":"2412.06559","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/processbench-identifying-process-errors-in#ran","syntology_url":"https://syntology.ai/paper/2412.06559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06559"}},"official":{"repos":["qwenlm/processbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/critical-tokens-matter-token-level","slug":"critical-tokens-matter-token-level","title":"Critical Tokens Matter: Token-Level Contrastive Estimation Enhances LLM's Reasoning Capability","date":"2024-11-29","arxiv_id":"2411.19943","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/critical-tokens-matter-token-level#ran","syntology_url":"https://syntology.ai/paper/2411.19943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19943"}},"official":{"repos":["chenzhiling9954/critical-tokens-matter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/training-and-evaluating-language-models-with","slug":"training-and-evaluating-language-models-with","title":"Training and Evaluating Language Models with Template-based Data Generation","date":"2024-11-27","arxiv_id":"2411.18104","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/training-and-evaluating-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2411.18104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18104"}},"official":{"repos":["iiis-ai/templatemath"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/preference-optimization-for-reasoning-with","slug":"preference-optimization-for-reasoning-with","title":"Preference Optimization for Reasoning with Pseudo Feedback","date":"2024-11-25","arxiv_id":"2411.16345","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/preference-optimization-for-reasoning-with#ran","syntology_url":"https://syntology.ai/paper/2411.16345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16345"}},"official":null}},{"url":"/paper/pspo-an-effective-process-supervised-policy","slug":"pspo-an-effective-process-supervised-policy","title":"PSPO*: An Effective Process-supervised Policy Optimization for Reasoning Alignment","date":"2024-11-18","arxiv_id":"2411.11681","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pspo-an-effective-process-supervised-policy#ran","syntology_url":"https://syntology.ai/paper/2411.11681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.11681"}},"official":{"repos":["direct-bit/pspo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stem-pom-evaluating-language-models-math","slug":"stem-pom-evaluating-language-models-math","title":"STEM-POM: Evaluating Language Models Math-Symbol Reasoning in Document Parsing","date":"2024-11-01","arxiv_id":"2411.00387","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stem-pom-evaluating-language-models-math#ran","syntology_url":"https://syntology.ai/paper/2411.00387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00387"}},"official":null}},{"url":"/paper/library-learning-doesn-t-the-curious-case-of","slug":"library-learning-doesn-t-the-curious-case-of","title":"Library Learning Doesn't: The Curious Case of the Single-Use \"Library\"","date":"2024-10-26","arxiv_id":"2410.20274","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/library-learning-doesn-t-the-curious-case-of#ran","syntology_url":"https://syntology.ai/paper/2410.20274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20274"}},"official":{"repos":["ikb-a/curious-case"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/assessing-the-creativity-of-llms-in-proposing","slug":"assessing-the-creativity-of-llms-in-proposing","title":"Assessing the Creativity of LLMs in Proposing Novel Solutions to Mathematical Problems","date":"2024-10-24","arxiv_id":"2410.18336","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/assessing-the-creativity-of-llms-in-proposing#ran","syntology_url":"https://syntology.ai/paper/2410.18336","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18336"}},"official":{"repos":["junyiye/creativemath"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unleashing-reasoning-capability-of-llms-via","slug":"unleashing-reasoning-capability-of-llms-via","title":"Unleashing Reasoning Capability of LLMs via Scalable Question Synthesis from Scratch","date":"2024-10-24","arxiv_id":"2410.18693","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unleashing-reasoning-capability-of-llms-via#ran","syntology_url":"https://syntology.ai/paper/2410.18693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18693"}},"official":{"repos":["yyding1/scalequest"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-leverage-demonstration-data-in","slug":"how-to-leverage-demonstration-data-in","title":"How to Leverage Demonstration Data in Alignment for Large Language Model? A Self-Imitation Learning Perspective","date":"2024-10-14","arxiv_id":"2410.10093","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/how-to-leverage-demonstration-data-in#ran","syntology_url":"https://syntology.ai/paper/2410.10093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10093"}},"official":{"repos":["tengxiao1/gsil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/comat-chain-of-mathematically-annotated","slug":"comat-chain-of-mathematically-annotated","title":"CoMAT: Chain of Mathematically Annotated Thought Improves Mathematical Reasoning","date":"2024-10-14","arxiv_id":"2410.10336","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/comat-chain-of-mathematically-annotated#ran","syntology_url":"https://syntology.ai/paper/2410.10336","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10336"}},"official":{"repos":["joshuaongg21/comat"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/omni-math-a-universal-olympiad-level","slug":"omni-math-a-universal-olympiad-level","title":"Omni-MATH: A Universal Olympiad Level Mathematic Benchmark For Large Language Models","date":"2024-10-10","arxiv_id":"2410.07985","repositories_listed":2,"syntology":{"n":19,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":19,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/omni-math-a-universal-olympiad-level#ran","syntology_url":"https://syntology.ai/paper/2410.07985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07985"}},"official":{"repos":["kbsdjames/omni-math","kbsdjames/omni-math-rule"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/teaching-inspired-integrated-prompting","slug":"teaching-inspired-integrated-prompting","title":"Teaching-Inspired Integrated Prompting Framework: A Novel Approach for Enhancing Reasoning in Large Language Models","date":"2024-10-10","arxiv_id":"2410.08068","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/teaching-inspired-integrated-prompting#ran","syntology_url":"https://syntology.ai/paper/2410.08068","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08068"}},"official":{"repos":["sallytan13/teaching-inspired-prompting"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mathcoder2-better-math-reasoning-from","slug":"mathcoder2-better-math-reasoning-from","title":"MathCoder2: Better Math Reasoning from Continued Pretraining on Model-translated Mathematical Code","date":"2024-10-10","arxiv_id":"2410.08196","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mathcoder2-better-math-reasoning-from#ran","syntology_url":"https://syntology.ai/paper/2410.08196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08196"}},"official":{"repos":["mathllm/mathcoder2"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/polymath-a-challenging-multi-modal","slug":"polymath-a-challenging-multi-modal","title":"Polymath: A Challenging Multi-modal Mathematical Reasoning Benchmark","date":"2024-10-06","arxiv_id":"2410.14702","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/polymath-a-challenging-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2410.14702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14702"}},"official":{"repos":["polymathbenchmark/PolyMATH"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llama-berry-pairwise-optimization-for-o1-like","slug":"llama-berry-pairwise-optimization-for-o1-like","title":"LLaMA-Berry: Pairwise Optimization for O1-like Olympiad-Level Mathematical Reasoning","date":"2024-10-03","arxiv_id":"2410.02884","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llama-berry-pairwise-optimization-for-o1-like#ran","syntology_url":"https://syntology.ai/paper/2410.02884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02884"}},"official":null}},{"url":"/paper/guided-stream-of-search-learning-to-better","slug":"guided-stream-of-search-learning-to-better","title":"Guided Stream of Search: Learning to Better Search with Language Models via Optimal Path Guidance","date":"2024-10-03","arxiv_id":"2410.02992","repositories_listed":1,"syntology":{"n":19,"n_ran":18,"n_constructed":0,"n_ran_checked":13,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":10,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/guided-stream-of-search-learning-to-better#ran","syntology_url":"https://syntology.ai/paper/2410.02992","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02992"}},"official":{"repos":["symoon11/guided-stream-of-search"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/openmathinstruct-2-accelerating-ai-for-math","slug":"openmathinstruct-2-accelerating-ai-for-math","title":"OpenMathInstruct-2: Accelerating AI for Math with Massive Open-Source Instruction Data","date":"2024-10-02","arxiv_id":"2410.01560","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/openmathinstruct-2-accelerating-ai-for-math#ran","syntology_url":"https://syntology.ai/paper/2410.01560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01560"}},"official":null}},{"url":"/paper/scheherazade-evaluating-chain-of-thought-math","slug":"scheherazade-evaluating-chain-of-thought-math","title":"Scheherazade: Evaluating Chain-of-Thought Math Reasoning in LLMs with Chain-of-Problems","date":"2024-09-30","arxiv_id":"2410.00151","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scheherazade-evaluating-chain-of-thought-math#ran","syntology_url":"https://syntology.ai/paper/2410.00151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00151"}},"official":{"repos":["yoshikitakashima/scheherazade-code-data"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pace-marrying-generalization-in-parameter","slug":"pace-marrying-generalization-in-parameter","title":"PACE: Marrying generalization in PArameter-efficient fine-tuning with Consistency rEgularization","date":"2024-09-25","arxiv_id":"2409.17137","repositories_listed":1,"syntology":{"n":18,"n_ran":8,"n_constructed":1,"n_ran_checked":7,"n_instrument":1,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/pace-marrying-generalization-in-parameter#ran","syntology_url":"https://syntology.ai/paper/2409.17137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.17137"}},"official":{"repos":["maxwellyaoni/pace"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":7,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to","slug":"cmm-math-a-chinese-multimodal-math-dataset-to","title":"CMM-Math: A Chinese Multimodal Math Dataset To Evaluate and Enhance the Mathematics Reasoning of Large Multimodal Models","date":"2024-09-04","arxiv_id":"2409.02834","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to#ran","syntology_url":"https://syntology.ai/paper/2409.02834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02834"}},"official":{"repos":["ecnu-icalk/educhat-math"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multimath-bridging-visual-and-mathematical","slug":"multimath-bridging-visual-and-mathematical","title":"MultiMath: Bridging Visual and Mathematical Reasoning for Large Language Models","date":"2024-08-30","arxiv_id":"2409.00147","repositories_listed":1,"syntology":{"n":18,"n_ran":17,"n_constructed":0,"n_ran_checked":9,"n_instrument":8,"n_unverified":1,"n_honours":0,"n_violates":3,"n_no_contract":6,"n_pointer_only":3,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 3 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multimath-bridging-visual-and-mathematical#ran","syntology_url":"https://syntology.ai/paper/2409.00147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00147"}},"official":{"repos":["pengshuai-rin/multimath"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/math-puma-progressive-upward-multimodal","slug":"math-puma-progressive-upward-multimodal","title":"Math-PUMA: Progressive Upward Multimodal Alignment to Enhance Mathematical Reasoning","date":"2024-08-16","arxiv_id":"2408.08640","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/math-puma-progressive-upward-multimodal#ran","syntology_url":"https://syntology.ai/paper/2408.08640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08640"}},"official":{"repos":["wwzhuang01/math-puma"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mathscape-evaluating-mllms-in-multimodal-math","slug":"mathscape-evaluating-mllms-in-multimodal-math","title":"MathScape: Evaluating MLLMs in multimodal Math Scenarios through a Hierarchical Benchmark","date":"2024-08-14","arxiv_id":"2408.07543","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathscape-evaluating-mllms-in-multimodal-math#ran","syntology_url":"https://syntology.ai/paper/2408.07543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07543"}},"official":{"repos":["PKU-Baichuan-MLSystemLab/MathScape"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/extend-model-merging-from-fine-tuned-to-pre","slug":"extend-model-merging-from-fine-tuned-to-pre","title":"Extend Model Merging from Fine-Tuned to Pre-Trained Large Language Models via Weight Disentanglement","date":"2024-08-06","arxiv_id":"2408.03092","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/extend-model-merging-from-fine-tuned-to-pre#ran","syntology_url":"https://syntology.ai/paper/2408.03092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03092"}},"official":{"repos":["yule-BUAA/MergeLLM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seallms-3-open-foundation-and-chat","slug":"seallms-3-open-foundation-and-chat","title":"SeaLLMs 3: Open Foundation and Chat Multilingual Large Language Models for Southeast Asian Languages","date":"2024-07-29","arxiv_id":"2407.19672","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/seallms-3-open-foundation-and-chat#ran","syntology_url":"https://syntology.ai/paper/2407.19672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.19672"}},"official":{"repos":["DAMO-NLP-SG/SeaExam"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/lora-pro-are-low-rank-adapters-properly","slug":"lora-pro-are-low-rank-adapters-properly","title":"LoRA-Pro: Are Low-Rank Adapters Properly Optimized?","date":"2024-07-25","arxiv_id":"2407.18242","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lora-pro-are-low-rank-adapters-properly#ran","syntology_url":"https://syntology.ai/paper/2407.18242","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18242"}},"official":{"repos":["mrflogs/LoRA-Pro"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/self-training-with-direct-preference","slug":"self-training-with-direct-preference","title":"Self-Training with Direct Preference Optimization Improves Chain-of-Thought Reasoning","date":"2024-07-25","arxiv_id":"2407.18248","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-training-with-direct-preference#ran","syntology_url":"https://syntology.ai/paper/2407.18248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18248"}},"official":{"repos":["tianduowang/dpo-st"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-llms-for-optimization-modeling","slug":"benchmarking-llms-for-optimization-modeling","title":"OptiBench Meets ReSocratic: Measure and Improve LLMs for Optimization Modeling","date":"2024-07-13","arxiv_id":"2407.09887","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-llms-for-optimization-modeling#ran","syntology_url":"https://syntology.ai/paper/2407.09887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09887"}},"official":{"repos":["yangzhch6/ReSocratic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-single-transformer-for-scalable-vision","slug":"a-single-transformer-for-scalable-vision","title":"SOLO: A Single Transformer for Scalable Vision-Language Modeling","date":"2024-07-08","arxiv_id":"2407.06438","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-single-transformer-for-scalable-vision#ran","syntology_url":"https://syntology.ai/paper/2407.06438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06438"}},"official":{"repos":["yangyi-chen/solo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/logicvista-multimodal-llm-logical-reasoning","slug":"logicvista-multimodal-llm-logical-reasoning","title":"LogicVista: Multimodal LLM Logical Reasoning Benchmark in Visual Contexts","date":"2024-07-06","arxiv_id":"2407.04973","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/logicvista-multimodal-llm-logical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2407.04973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04973"}},"official":{"repos":["yijia-xiao/logicvista"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/smart-vision-language-reasoners","slug":"smart-vision-language-reasoners","title":"Smart Vision-Language Reasoners","date":"2024-07-05","arxiv_id":"2407.04212","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/smart-vision-language-reasoners#ran","syntology_url":"https://syntology.ai/paper/2407.04212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04212"}},"official":{"repos":["smarter-vlm/smarter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/theoremllama-transforming-general-purpose","slug":"theoremllama-transforming-general-purpose","title":"TheoremLlama: Transforming General-Purpose LLMs into Lean4 Experts","date":"2024-07-03","arxiv_id":"2407.03203","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/theoremllama-transforming-general-purpose#ran","syntology_url":"https://syntology.ai/paper/2407.03203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03203"}},"official":{"repos":["RickySkywalker/TheoremLlama"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"a2663f827b099d8f565bbc902669d6f641a58dfc76fbe653aa172608ffad3402","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}