{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/math/papers/3","list_of":"/task/math","task":"Math","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":16,"rows_per_page":100,"rows":[201,300],"of":1596,"counts":{"archive_papers_tagged":1596,"with_a_code_link":765,"where_syntology_ran_a_sample":349,"not_listed_spam_title":0,"listed":1596,"listed_where_code_ran":349,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/math","prev":"/task/math/papers/2","next":"/task/math/papers/4","papers":[{"url":"/paper/thinkless-llm-learns-when-to-think","slug":"thinkless-llm-learns-when-to-think","title":"Thinkless: LLM Learns When to Think","date":"2025-05-19","arxiv_id":"2505.13379","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/thinkless-llm-learns-when-to-think#ran","syntology_url":"https://syntology.ai/paper/2505.13379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13379"}},"official":{"repos":["vainf/thinkless"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/warm-up-before-you-train-unlocking-general","slug":"warm-up-before-you-train-unlocking-general","title":"Warm Up Before You Train: Unlocking General Reasoning in Resource-Constrained Settings","date":"2025-05-19","arxiv_id":"2505.13718","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-rl-training-for-reasoning-models","slug":"efficient-rl-training-for-reasoning-models","title":"Efficient RL Training for Reasoning Models via Length-Aware Optimization","date":"2025-05-18","arxiv_id":"2505.12284","repositories_listed":1,"syntology":null},{"url":"/paper/marge-improving-math-reasoning-for-llms-with","slug":"marge-improving-math-reasoning-for-llms-with","title":"MARGE: Improving Math Reasoning for LLMs with Guided Exploration","date":"2025-05-18","arxiv_id":"2505.12500","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/marge-improving-math-reasoning-for-llms-with#ran","syntology_url":"https://syntology.ai/paper/2505.12500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12500"}},"official":{"repos":["georgao35/marge"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","unlocated"]}}},{"url":"/paper/synthetic-data-rl-task-definition-is-all-you","slug":"synthetic-data-rl-task-definition-is-all-you","title":"Synthetic Data RL: Task Definition Is All You Need","date":"2025-05-18","arxiv_id":"2505.17063","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/synthetic-data-rl-task-definition-is-all-you#ran","syntology_url":"https://syntology.ai/paper/2505.17063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17063"}},"official":{"repos":["gydpku/data_synthesis_rl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/halo-hierarchical-autonomous-logic-oriented","slug":"halo-hierarchical-autonomous-logic-oriented","title":"HALO: Hierarchical Autonomous Logic-Oriented Orchestration for Multi-Agent LLM Systems","date":"2025-05-17","arxiv_id":"2505.13516","repositories_listed":1,"syntology":null},{"url":"/paper/hardmath2-a-benchmark-for-applied-mathematics","slug":"hardmath2-a-benchmark-for-applied-mathematics","title":"HARDMath2: A Benchmark for Applied Mathematics Built by Students as Part of a Graduate Class","date":"2025-05-17","arxiv_id":"2505.11774","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11225","slug":"2505-11225","title":"HAPO: Training Language Models to Reason Concisely via History-Aware Policy Optimization","date":"2025-05-16","arxiv_id":"2505.11225","repositories_listed":1,"syntology":null},{"url":"/paper/medcasereasoning-evaluating-and-learning","slug":"medcasereasoning-evaluating-and-learning","title":"MedCaseReasoning: Evaluating and learning diagnostic reasoning from clinical case reports","date":"2025-05-16","arxiv_id":"2505.11733","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-aha-toward-systematic-meta-abilities","slug":"beyond-aha-toward-systematic-meta-abilities","title":"Beyond 'Aha!': Toward Systematic Meta-Abilities Alignment in Large Reasoning Models","date":"2025-05-15","arxiv_id":"2505.10554","repositories_listed":1,"syntology":null},{"url":"/paper/mathcoder-vl-bridging-vision-and-code-for","slug":"mathcoder-vl-bridging-vision-and-code-for","title":"MathCoder-VL: Bridging Vision and Code for Enhanced Multimodal Mathematical Reasoning","date":"2025-05-15","arxiv_id":"2505.10557","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathcoder-vl-bridging-vision-and-code-for#ran","syntology_url":"https://syntology.ai/paper/2505.10557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10557"}},"official":{"repos":["mathllm/mathcoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-a-deeper-understanding-of-reasoning","slug":"towards-a-deeper-understanding-of-reasoning","title":"Towards a Deeper Understanding of Reasoning Capabilities in Large Language Models","date":"2025-05-15","arxiv_id":"2505.10543","repositories_listed":1,"syntology":null},{"url":"/paper/pt-moe-an-efficient-finetuning-framework-for","slug":"pt-moe-an-efficient-finetuning-framework-for","title":"PT-MoE: An Efficient Finetuning Framework for Integrating Mixture-of-Experts into Prompt Tuning","date":"2025-05-14","arxiv_id":"2505.09519","repositories_listed":1,"syntology":null},{"url":"/paper/kalman-filter-enhanced-grpo-for-reinforcement","slug":"kalman-filter-enhanced-grpo-for-reinforcement","title":"Kalman Filter Enhanced GRPO for Reinforcement Learning-Based Language Model Reasoning","date":"2025-05-12","arxiv_id":"2505.07527","repositories_listed":1,"syntology":null},{"url":"/paper/rewriting-pre-training-data-boosts-llm","slug":"rewriting-pre-training-data-boosts-llm","title":"Rewriting Pre-Training Data Boosts LLM Performance in Math and Code","date":"2025-05-05","arxiv_id":"2505.02881","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rewriting-pre-training-data-boosts-llm#ran","syntology_url":"https://syntology.ai/paper/2505.02881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02881"}},"official":{"repos":["rioyokotalab/swallow-code-math"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rm-r1-reward-modeling-as-reasoning","slug":"rm-r1-reward-modeling-as-reasoning","title":"RM-R1: Reward Modeling as Reasoning","date":"2025-05-05","arxiv_id":"2505.02387","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/rm-r1-reward-modeling-as-reasoning#ran","syntology_url":"https://syntology.ai/paper/2505.02387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02387"}},"official":{"repos":["rm-r1-uiuc/rm-r1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tutorgym-a-testbed-for-evaluating-ai-agents","slug":"tutorgym-a-testbed-for-evaluating-ai-agents","title":"TutorGym: A Testbed for Evaluating AI Agents as Tutors and Students","date":"2025-05-02","arxiv_id":"2505.01563","repositories_listed":1,"syntology":null},{"url":"/paper/deepcritic-deliberate-critique-with-large","slug":"deepcritic-deliberate-critique-with-large","title":"DeepCritic: Deliberate Critique with Large Language Models","date":"2025-05-01","arxiv_id":"2505.00662","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepcritic-deliberate-critique-with-large#ran","syntology_url":"https://syntology.ai/paper/2505.00662","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.00662"}},"official":{"repos":["rucbm/deepcritic"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/nemo-inspector-a-visualization-tool-for-llm","slug":"nemo-inspector-a-visualization-tool-for-llm","title":"NeMo-Inspector: A Visualization Tool for LLM Generation Analysis","date":"2025-05-01","arxiv_id":"2505.00903","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-reasoning-in-large","slug":"reinforcement-learning-for-reasoning-in-large","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","date":"2025-04-29","arxiv_id":"2504.20571","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":9,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/reinforcement-learning-for-reasoning-in-large#ran","syntology_url":"https://syntology.ai/paper/2504.20571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20571"}},"official":{"repos":["ypwang61/one-shot-rlvr"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-reasoning-for-llms-through","slug":"efficient-reasoning-for-llms-through","title":"Efficient Reasoning for LLMs through Speculative Chain-of-Thought","date":"2025-04-27","arxiv_id":"2504.19095","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-reasoning-for-llms-through#ran","syntology_url":"https://syntology.ai/paper/2504.19095","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.19095"}},"official":{"repos":["jikai0wang/speculative_cot"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/2504-18589","slug":"2504-18589","title":"Benchmarking Multimodal Mathematical Reasoning with Explicit Visual Dependency","date":"2025-04-24","arxiv_id":"2504.18589","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-on-prompt-compression-for","slug":"an-empirical-study-on-prompt-compression-for","title":"An Empirical Study on Prompt Compression for Large Language Models","date":"2025-04-24","arxiv_id":"2505.00019","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-on-prompt-compression-for#ran","syntology_url":"https://syntology.ai/paper/2505.00019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.00019"}},"official":{"repos":["3DAgentWorld/Toolkit-for-Prompt-Compression"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aimo-2-winning-solution-building-state-of-the","slug":"aimo-2-winning-solution-building-state-of-the","title":"AIMO-2 Winning Solution: Building State-of-the-Art Mathematical Reasoning Models with OpenMathReasoning dataset","date":"2025-04-23","arxiv_id":"2504.16891","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/aimo-2-winning-solution-building-state-of-the#ran","syntology_url":"https://syntology.ai/paper/2504.16891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16891"}},"official":null}},{"url":"/paper/process-reward-models-that-think","slug":"process-reward-models-that-think","title":"Process Reward Models That Think","date":"2025-04-23","arxiv_id":"2504.16828","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/process-reward-models-that-think#ran","syntology_url":"https://syntology.ai/paper/2504.16828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16828"}},"official":{"repos":["mukhal/thinkprm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dianjin-r1-evaluating-and-enhancing-financial","slug":"dianjin-r1-evaluating-and-enhancing-financial","title":"DianJin-R1: Evaluating and Enhancing Financial Reasoning in Large Language Models","date":"2025-04-22","arxiv_id":"2504.15716","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dianjin-r1-evaluating-and-enhancing-financial#ran","syntology_url":"https://syntology.ai/paper/2504.15716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15716"}},"official":{"repos":["aliyun/qwen-dianjin"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-early-exit-in-reasoning-models","slug":"dynamic-early-exit-in-reasoning-models","title":"Dynamic Early Exit in Reasoning Models","date":"2025-04-22","arxiv_id":"2504.15895","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-early-exit-in-reasoning-models#ran","syntology_url":"https://syntology.ai/paper/2504.15895","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15895"}},"official":{"repos":["iie-ycx/deer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-judges-as-evaluators-the-jetts","slug":"evaluating-judges-as-evaluators-the-jetts","title":"Evaluating Judges as Evaluators: The JETTS Benchmark of LLM-as-Judges as Test-Time Scaling Evaluators","date":"2025-04-21","arxiv_id":"2504.15253","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-judges-as-evaluators-the-jetts#ran","syntology_url":"https://syntology.ai/paper/2504.15253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15253"}},"official":{"repos":["salesforceairesearch/jetts-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-reason-under-off-policy-guidance","slug":"learning-to-reason-under-off-policy-guidance","title":"Learning to Reason under Off-Policy Guidance","date":"2025-04-21","arxiv_id":"2504.14945","repositories_listed":1,"syntology":null},{"url":"/paper/roll-the-dice-look-before-you-leap-going","slug":"roll-the-dice-look-before-you-leap-going","title":"Roll the dice & look before you leap: Going beyond the creative limits of next-token prediction","date":"2025-04-21","arxiv_id":"2504.15266","repositories_listed":1,"syntology":null},{"url":"/paper/stop-summation-min-form-credit-assignment-is","slug":"stop-summation-min-form-credit-assignment-is","title":"Stop Summation: Min-Form Credit Assignment Is All Process Reward Model Needs for Reasoning","date":"2025-04-21","arxiv_id":"2504.15275","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stop-summation-min-form-credit-assignment-is#ran","syntology_url":"https://syntology.ai/paper/2504.15275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15275"}},"official":{"repos":["cjreinforce/pure"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-from-human-feedback-4","slug":"reinforcement-learning-from-human-feedback-4","title":"Reinforcement Learning from Human Feedback","date":"2025-04-16","arxiv_id":"2504.12501","repositories_listed":1,"syntology":null},{"url":"/paper/fine-tuning-large-language-models-on-quantum","slug":"fine-tuning-large-language-models-on-quantum","title":"Fine-Tuning Large Language Models on Quantum Optimization Problems for Circuit Generation","date":"2025-04-15","arxiv_id":"2504.11109","repositories_listed":1,"syntology":null},{"url":"/paper/retool-reinforcement-learning-for-strategic","slug":"retool-reinforcement-learning-for-strategic","title":"ReTool: Reinforcement Learning for Strategic Tool Use in LLMs","date":"2025-04-15","arxiv_id":"2504.11536","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-process-reward-model-training-via","slug":"efficient-process-reward-model-training-via","title":"Efficient Process Reward Model Training via Active Learning","date":"2025-04-14","arxiv_id":"2504.10559","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-process-reward-model-training-via#ran","syntology_url":"https://syntology.ai/paper/2504.10559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10559"}},"official":{"repos":["sail-sg/activeprm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/m1-towards-scalable-test-time-compute-with","slug":"m1-towards-scalable-test-time-compute-with","title":"M1: Towards Scalable Test-Time Compute with Mamba Reasoning Models","date":"2025-04-14","arxiv_id":"2504.10449","repositories_listed":1,"syntology":null},{"url":"/paper/the-jailbreak-tax-how-useful-are-your","slug":"the-jailbreak-tax-how-useful-are-your","title":"The Jailbreak Tax: How Useful are Your Jailbreak Outputs?","date":"2025-04-14","arxiv_id":"2504.10694","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-jailbreak-tax-how-useful-are-your#ran","syntology_url":"https://syntology.ai/paper/2504.10694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10694"}},"official":{"repos":["ethz-spylab/jailbreak-tax"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/syzygy-of-thoughts-improving-llm-cot-with-the","slug":"syzygy-of-thoughts-improving-llm-cot-with-the","title":"Syzygy of Thoughts: Improving LLM CoT with the Minimal Free Resolution","date":"2025-04-13","arxiv_id":"2504.09566","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-cheatsheet-test-time-learning-with","slug":"dynamic-cheatsheet-test-time-learning-with","title":"Dynamic Cheatsheet: Test-Time Learning with Adaptive Memory","date":"2025-04-10","arxiv_id":"2504.07952","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-cheatsheet-test-time-learning-with#ran","syntology_url":"https://syntology.ai/paper/2504.07952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07952"}},"official":{"repos":["suzgunmirac/dynamic-cheatsheet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/task-circuit-quantization-leveraging","slug":"task-circuit-quantization-leveraging","title":"Task-Circuit Quantization: Leveraging Knowledge Localization and Interpretability for Compression","date":"2025-04-10","arxiv_id":"2504.07389","repositories_listed":1,"syntology":null},{"url":"/paper/mdk12-bench-a-multi-discipline-benchmark-for-1","slug":"mdk12-bench-a-multi-discipline-benchmark-for-1","title":"MDK12-Bench: A Multi-Discipline Benchmark for Evaluating Reasoning in Multimodal Large Language Models","date":"2025-04-08","arxiv_id":"2504.05782","repositories_listed":1,"syntology":null},{"url":"/paper/right-question-is-already-half-the-answer","slug":"right-question-is-already-half-the-answer","title":"Right Question is Already Half the Answer: Fully Unsupervised LLM Reasoning Incentivization","date":"2025-04-08","arxiv_id":"2504.05812","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/right-question-is-already-half-the-answer#ran","syntology_url":"https://syntology.ai/paper/2504.05812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05812"}},"official":{"repos":["qingyangzhang/empo"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/beyond-single-turn-a-survey-on-multi-turn","slug":"beyond-single-turn-a-survey-on-multi-turn","title":"Beyond Single-Turn: A Survey on Multi-Turn Interactions with Large Language Models","date":"2025-04-07","arxiv_id":"2504.04717","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-reinforcement-finetuning-via","slug":"efficient-reinforcement-finetuning-via","title":"Efficient Reinforcement Finetuning via Adaptive Curriculum Learning","date":"2025-04-07","arxiv_id":"2504.05520","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/efficient-reinforcement-finetuning-via#ran","syntology_url":"https://syntology.ai/paper/2504.05520","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05520"}},"official":{"repos":["uscnlp-lime/verl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/quantization-hurts-reasoning-an-empirical","slug":"quantization-hurts-reasoning-an-empirical","title":"Quantization Hurts Reasoning? An Empirical Study on Quantized Reasoning Models","date":"2025-04-07","arxiv_id":"2504.04823","repositories_listed":1,"syntology":null},{"url":"/paper/large-vision-language-models-are-unsupervised","slug":"large-vision-language-models-are-unsupervised","title":"Large (Vision) Language Models are Unsupervised In-Context Learners","date":"2025-04-03","arxiv_id":"2504.02349","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/large-vision-language-models-are-unsupervised#ran","syntology_url":"https://syntology.ai/paper/2504.02349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02349"}},"official":{"repos":["mlbio-epfl/joint-inference"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/megamath-pushing-the-limits-of-open-math","slug":"megamath-pushing-the-limits-of-open-math","title":"MegaMath: Pushing the Limits of Open Math Corpora","date":"2025-04-03","arxiv_id":"2504.02807","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/megamath-pushing-the-limits-of-open-math#ran","syntology_url":"https://syntology.ai/paper/2504.02807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02807"}},"official":{"repos":["llm360/megamath"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/blendergym-benchmarking-foundational-model","slug":"blendergym-benchmarking-foundational-model","title":"BlenderGym: Benchmarking Foundational Model Systems for Graphics Editing","date":"2025-04-02","arxiv_id":"2504.01786","repositories_listed":1,"syntology":null},{"url":"/paper/an-extrapolated-and-provably-convergent","slug":"an-extrapolated-and-provably-convergent","title":"An extrapolated and provably convergent algorithm for nonlinear matrix decomposition with the ReLU function","date":"2025-03-31","arxiv_id":"2503.23832","repositories_listed":1,"syntology":null},{"url":"/paper/entropy-based-adaptive-weighting-for-self","slug":"entropy-based-adaptive-weighting-for-self","title":"Entropy-Based Adaptive Weighting for Self-Training","date":"2025-03-31","arxiv_id":"2503.23913","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/entropy-based-adaptive-weighting-for-self#ran","syntology_url":"https://syntology.ai/paper/2503.23913","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23913"}},"official":{"repos":["mandyyyyii/east"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inference-time-scaling-for-complex-tasks","slug":"inference-time-scaling-for-complex-tasks","title":"Inference-Time Scaling for Complex Tasks: Where We Stand and What Lies Ahead","date":"2025-03-31","arxiv_id":"2504.00294","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inference-time-scaling-for-complex-tasks#ran","syntology_url":"https://syntology.ai/paper/2504.00294","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.00294"}},"official":null}},{"url":"/paper/torl-scaling-tool-integrated-rl","slug":"torl-scaling-tool-integrated-rl","title":"ToRL: Scaling Tool-Integrated RL","date":"2025-03-30","arxiv_id":"2503.23383","repositories_listed":1,"syntology":null},{"url":"/paper/cppo-accelerating-the-training-of-group","slug":"cppo-accelerating-the-training-of-group","title":"CPPO: Accelerating the Training of Group Relative Policy Optimization-Based Reasoning Models","date":"2025-03-28","arxiv_id":"2503.22342","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/cppo-accelerating-the-training-of-group#ran","syntology_url":"https://syntology.ai/paper/2503.22342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.22342"}},"official":{"repos":["lzhxmu/cppo"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-reason-for-long-form-story","slug":"learning-to-reason-for-long-form-story","title":"Learning to Reason for Long-Form Story Generation","date":"2025-03-28","arxiv_id":"2503.22828","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":1,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-reason-for-long-form-story#ran","syntology_url":"https://syntology.ai/paper/2503.22828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.22828"}},"official":{"repos":["Alex-Gurung/ReasoningNCP"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/questbench-can-llms-ask-the-right-question-to","slug":"questbench-can-llms-ask-the-right-question-to","title":"QuestBench: Can LLMs ask the right question to acquire information in reasoning tasks?","date":"2025-03-28","arxiv_id":"2503.22674","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/questbench-can-llms-ask-the-right-question-to#ran","syntology_url":"https://syntology.ai/paper/2503.22674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.22674"}},"official":{"repos":["google-deepmind/questbench"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/effective-skill-unlearning-through","slug":"effective-skill-unlearning-through","title":"Effective Skill Unlearning through Intervention and Abstention","date":"2025-03-27","arxiv_id":"2503.21730","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/effective-skill-unlearning-through#ran","syntology_url":"https://syntology.ai/paper/2503.21730","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21730"}},"official":{"repos":["trustworthy-ml-lab/effective_skill_unlearning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/thinkedit-interpretable-weight-editing-to","slug":"thinkedit-interpretable-weight-editing-to","title":"ThinkEdit: Interpretable Weight Editing to Mitigate Overly Short Thinking in Reasoning Models","date":"2025-03-27","arxiv_id":"2503.22048","repositories_listed":1,"syntology":null},{"url":"/paper/logquant-log-distributed-2-bit-quantization","slug":"logquant-log-distributed-2-bit-quantization","title":"LogQuant: Log-Distributed 2-Bit Quantization of KV Cache with Superior Accuracy Preservation","date":"2025-03-25","arxiv_id":"2503.19950","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-to-learn-from-latent-thoughts","slug":"reasoning-to-learn-from-latent-thoughts","title":"Reasoning to Learn from Latent Thoughts","date":"2025-03-24","arxiv_id":"2503.18866","repositories_listed":1,"syntology":{"n":20,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":4,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/reasoning-to-learn-from-latent-thoughts#ran","syntology_url":"https://syntology.ai/paper/2503.18866","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18866"}},"official":null}},{"url":"/paper/simplerl-zoo-investigating-and-taming-zero","slug":"simplerl-zoo-investigating-and-taming-zero","title":"SimpleRL-Zoo: Investigating and Taming Zero Reinforcement Learning for Open Base Models in the Wild","date":"2025-03-24","arxiv_id":"2503.18892","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simplerl-zoo-investigating-and-taming-zero#ran","syntology_url":"https://syntology.ai/paper/2503.18892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18892"}},"official":null}},{"url":"/paper/agentrxiv-towards-collaborative-autonomous","slug":"agentrxiv-towards-collaborative-autonomous","title":"AgentRxiv: Towards Collaborative Autonomous Research","date":"2025-03-23","arxiv_id":"2503.18102","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/agentrxiv-towards-collaborative-autonomous#ran","syntology_url":"https://syntology.ai/paper/2503.18102","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18102"}},"official":null}},{"url":"/paper/lost-in-cultural-translation-do-llms-struggle","slug":"lost-in-cultural-translation-do-llms-struggle","title":"Lost in Cultural Translation: Do LLMs Struggle with Math Across Cultural Contexts?","date":"2025-03-23","arxiv_id":"2503.18018","repositories_listed":1,"syntology":null},{"url":"/paper/chatbench-from-static-benchmarks-to-human-ai","slug":"chatbench-from-static-benchmarks-to-human-ai","title":"ChatBench: From Static Benchmarks to Human-AI Evaluation","date":"2025-03-22","arxiv_id":"2504.07114","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chatbench-from-static-benchmarks-to-human-ai#ran","syntology_url":"https://syntology.ai/paper/2504.07114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07114"}},"official":{"repos":["serinachang5/interactive-eval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/fastcurl-curriculum-reinforcement-learning","slug":"fastcurl-curriculum-reinforcement-learning","title":"FastCuRL: Curriculum Reinforcement Learning with Progressive Context Extension for Efficient Training R1-like Reasoning Models","date":"2025-03-21","arxiv_id":"2503.17287","repositories_listed":1,"syntology":null},{"url":"/paper/burtorch-revisiting-training-from-first","slug":"burtorch-revisiting-training-from-first","title":"BurTorch: Revisiting Training from First Principles by Coupling Autodiff, Math Optimization, and Systems","date":"2025-03-18","arxiv_id":"2503.13795","repositories_listed":1,"syntology":null},{"url":"/paper/exaone-deep-reasoning-enhanced-language","slug":"exaone-deep-reasoning-enhanced-language","title":"EXAONE Deep: Reasoning Enhanced Language Models","date":"2025-03-16","arxiv_id":"2503.12524","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exaone-deep-reasoning-enhanced-language#ran","syntology_url":"https://syntology.ai/paper/2503.12524","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.12524"}},"official":null}},{"url":"/paper/light-r1-curriculum-sft-dpo-and-rl-for-long","slug":"light-r1-curriculum-sft-dpo-and-rl-for-long","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","date":"2025-03-13","arxiv_id":"2503.10460","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/light-r1-curriculum-sft-dpo-and-rl-for-long#ran","syntology_url":"https://syntology.ai/paper/2503.10460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10460"}},"official":{"repos":["qihoo360/light-r1"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/stepmathagent-a-step-wise-agent-for","slug":"stepmathagent-a-step-wise-agent-for","title":"StepMathAgent: A Step-Wise Agent for Evaluating Mathematical Processes through Tree-of-Error","date":"2025-03-13","arxiv_id":"2503.10105","repositories_listed":1,"syntology":null},{"url":"/paper/visualwebinstruct-scaling-up-multimodal","slug":"visualwebinstruct-scaling-up-multimodal","title":"VisualWebInstruct: Scaling up Multimodal Instruction Data through Web Search","date":"2025-03-13","arxiv_id":"2503.10582","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/visualwebinstruct-scaling-up-multimodal#ran","syntology_url":"https://syntology.ai/paper/2503.10582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10582"}},"official":null}},{"url":"/paper/evaltree-profiling-language-model-weaknesses","slug":"evaltree-profiling-language-model-weaknesses","title":"EvalTree: Profiling Language Model Weaknesses via Hierarchical Capability Trees","date":"2025-03-11","arxiv_id":"2503.08893","repositories_listed":1,"syntology":null},{"url":"/paper/vision-r1-incentivizing-reasoning-capability","slug":"vision-r1-incentivizing-reasoning-capability","title":"Vision-R1: Incentivizing Reasoning Capability in Multimodal Large Language Models","date":"2025-03-09","arxiv_id":"2503.06749","repositories_listed":1,"syntology":null},{"url":"/paper/promptcot-synthesizing-olympiad-level","slug":"promptcot-synthesizing-olympiad-level","title":"PromptCoT: Synthesizing Olympiad-level Problems for Mathematical Reasoning in Large Language Models","date":"2025-03-04","arxiv_id":"2503.02324","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/promptcot-synthesizing-olympiad-level#ran","syntology_url":"https://syntology.ai/paper/2503.02324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.02324"}},"official":{"repos":["zhaoxlpku/promptcot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-an-llm-is-apprehensive-about-its-answers","slug":"when-an-llm-is-apprehensive-about-its-answers","title":"When an LLM is apprehensive about its answers -- and when its uncertainty is justified","date":"2025-03-03","arxiv_id":"2503.01688","repositories_listed":1,"syntology":null},{"url":"/paper/finereason-evaluating-and-improving-llms","slug":"finereason-evaluating-and-improving-llms","title":"FINEREASON: Evaluating and Improving LLMs' Deliberate Reasoning through Reflective Puzzle Solving","date":"2025-02-27","arxiv_id":"2502.20238","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/finereason-evaluating-and-improving-llms#ran","syntology_url":"https://syntology.ai/paper/2502.20238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20238"}},"official":{"repos":["DAMO-NLP-SG/FineReason"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/self-training-elicits-concise-reasoning-in","slug":"self-training-elicits-concise-reasoning-in","title":"Self-Training Elicits Concise Reasoning in Large Language Models","date":"2025-02-27","arxiv_id":"2502.20122","repositories_listed":1,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-training-elicits-concise-reasoning-in#ran","syntology_url":"https://syntology.ai/paper/2502.20122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20122"}},"official":{"repos":["tergelmunkhbat/concise-reasoning"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-large-language-models-detect-errors-in","slug":"can-large-language-models-detect-errors-in","title":"Can Large Language Models Detect Errors in Long Chain-of-Thought Reasoning?","date":"2025-02-26","arxiv_id":"2502.19361","repositories_listed":1,"syntology":null},{"url":"/paper/nexus-a-lightweight-and-scalable-multi-agent","slug":"nexus-a-lightweight-and-scalable-multi-agent","title":"Nexus: A Lightweight and Scalable Multi-Agent Framework for Complex Tasks Automation","date":"2025-02-26","arxiv_id":"2502.19091","repositories_listed":1,"syntology":null},{"url":"/paper/big-math-a-large-scale-high-quality-math","slug":"big-math-a-large-scale-high-quality-math","title":"Big-Math: A Large-Scale, High-Quality Math Dataset for Reinforcement Learning in Language Models","date":"2025-02-24","arxiv_id":"2502.17387","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/big-math-a-large-scale-high-quality-math#ran","syntology_url":"https://syntology.ai/paper/2502.17387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17387"}},"official":{"repos":["synthlabsai/big-math"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/from-euler-to-ai-unifying-formulas-for","slug":"from-euler-to-ai-unifying-formulas-for","title":"From Euler to AI: Unifying Formulas for Mathematical Constants","date":"2025-02-24","arxiv_id":"2502.17533","repositories_listed":1,"syntology":null},{"url":"/paper/linguistic-generalizability-of-test-time","slug":"linguistic-generalizability-of-test-time","title":"Linguistic Generalizability of Test-Time Scaling in Mathematical Reasoning","date":"2025-02-24","arxiv_id":"2502.17407","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/linguistic-generalizability-of-test-time#ran","syntology_url":"https://syntology.ai/paper/2502.17407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17407"}},"official":{"repos":["gauss5930/mclm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/forgotten-polygons-multimodal-large-language","slug":"forgotten-polygons-multimodal-large-language","title":"Forgotten Polygons: Multimodal Large Language Models are Shape-Blind","date":"2025-02-21","arxiv_id":"2502.15969","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/forgotten-polygons-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2502.15969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15969"}},"official":{"repos":["rsinghlab/shape-blind"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/the-relationship-between-reasoning-and","slug":"the-relationship-between-reasoning-and","title":"The Relationship Between Reasoning and Performance in Large Language Models -- o3 (mini) Thinks Harder, Not Longer","date":"2025-02-21","arxiv_id":"2502.15631","repositories_listed":1,"syntology":null},{"url":"/paper/cer-confidence-enhanced-reasoning-in-llms","slug":"cer-confidence-enhanced-reasoning-in-llms","title":"CER: Confidence Enhanced Reasoning in LLMs","date":"2025-02-20","arxiv_id":"2502.14634","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":0,"n_instrument":9,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":15,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 9 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/cer-confidence-enhanced-reasoning-in-llms#ran","syntology_url":"https://syntology.ai/paper/2502.14634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14634"}},"official":{"repos":["sharif-ml-lab/CER"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/earlier-tokens-contribute-more-learning","slug":"earlier-tokens-contribute-more-learning","title":"Earlier Tokens Contribute More: Learning Direct Preference Optimization From Temporal Decay Perspective","date":"2025-02-20","arxiv_id":"2502.14340","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/earlier-tokens-contribute-more-learning#ran","syntology_url":"https://syntology.ai/paper/2502.14340","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14340"}},"official":{"repos":["lotusrc/d2po"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gate-graph-based-adaptive-tool-evolution","slug":"gate-graph-based-adaptive-tool-evolution","title":"GATE: Graph-based Adaptive Tool Evolution Across Diverse Tasks","date":"2025-02-20","arxiv_id":"2502.14848","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-get-your-llm-to-generate-challenging","slug":"how-to-get-your-llm-to-generate-challenging","title":"How to Get Your LLM to Generate Challenging Problems for Evaluation","date":"2025-02-20","arxiv_id":"2502.14678","repositories_listed":1,"syntology":null},{"url":"/paper/s-test-time-scaling-for-code-generation","slug":"s-test-time-scaling-for-code-generation","title":"S*: Test Time Scaling for Code Generation","date":"2025-02-20","arxiv_id":"2502.14382","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/s-test-time-scaling-for-code-generation#ran","syntology_url":"https://syntology.ai/paper/2502.14382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14382"}},"official":{"repos":["novasky-ai/skythought"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reasoning-with-reinforced-functional-token","slug":"reasoning-with-reinforced-functional-token","title":"Reasoning with Reinforced Functional Token Tuning","date":"2025-02-19","arxiv_id":"2502.13389","repositories_listed":1,"syntology":null},{"url":"/paper/sift-grounding-llm-reasoning-in-contexts-via","slug":"sift-grounding-llm-reasoning-in-contexts-via","title":"SIFT: Grounding LLM Reasoning in Contexts via Stickers","date":"2025-02-19","arxiv_id":"2502.14922","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sift-grounding-llm-reasoning-in-contexts-via#ran","syntology_url":"https://syntology.ai/paper/2502.14922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14922"}},"official":{"repos":["zhijie-group/sift"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/treecut-a-synthetic-unanswerable-math-word","slug":"treecut-a-synthetic-unanswerable-math-word","title":"TreeCut: A Synthetic Unanswerable Math Word Problem Dataset for LLM Hallucination Evaluation","date":"2025-02-19","arxiv_id":"2502.13442","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/treecut-a-synthetic-unanswerable-math-word#ran","syntology_url":"https://syntology.ai/paper/2502.13442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13442"}},"official":{"repos":["j-bagel/treecut-math"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/s-2-r-teaching-llms-to-self-verify-and-self","slug":"s-2-r-teaching-llms-to-self-verify-and-self","title":"S$^2$R: Teaching LLMs to Self-verify and Self-correct via Reinforcement Learning","date":"2025-02-18","arxiv_id":"2502.12853","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/s-2-r-teaching-llms-to-self-verify-and-self#ran","syntology_url":"https://syntology.ai/paper/2502.12853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.12853"}},"official":{"repos":["nineabyss/s2r"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/code-vision-evaluating-multimodal-llms-logic","slug":"code-vision-evaluating-multimodal-llms-logic","title":"Code-Vision: Evaluating Multimodal LLMs Logic Understanding and Code Generation Capabilities","date":"2025-02-17","arxiv_id":"2502.11829","repositories_listed":1,"syntology":null},{"url":"/paper/thinking-preference-optimization","slug":"thinking-preference-optimization","title":"Thinking Preference Optimization","date":"2025-02-17","arxiv_id":"2502.13173","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/thinking-preference-optimization#ran","syntology_url":"https://syntology.ai/paper/2502.13173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13173"}},"official":{"repos":["uservan/ThinkPO"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/uncovering-the-impact-of-chain-of-thought","slug":"uncovering-the-impact-of-chain-of-thought","title":"Uncovering the Impact of Chain-of-Thought Reasoning for Direct Preference Optimization: Lessons from Text-to-SQL","date":"2025-02-17","arxiv_id":"2502.11656","repositories_listed":1,"syntology":null},{"url":"/paper/warmup-distill-bridge-the-distribution","slug":"warmup-distill-bridge-the-distribution","title":"Warmup-Distill: Bridge the Distribution Mismatch between Teacher and Student before Knowledge Distillation","date":"2025-02-17","arxiv_id":"2502.11766","repositories_listed":1,"syntology":null},{"url":"/paper/don-t-get-lost-in-the-trees-streamlining-llm","slug":"don-t-get-lost-in-the-trees-streamlining-llm","title":"Don't Get Lost in the Trees: Streamlining LLM Reasoning by Overcoming Tree Search Exploration Pitfalls","date":"2025-02-16","arxiv_id":"2502.11183","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/don-t-get-lost-in-the-trees-streamlining-llm#ran","syntology_url":"https://syntology.ai/paper/2502.11183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11183"}},"official":{"repos":["soistesimmer/fetch"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/dyve-thinking-fast-and-slow-for-dynamic","slug":"dyve-thinking-fast-and-slow-for-dynamic","title":"Dyve: Thinking Fast and Slow for Dynamic Process Verification","date":"2025-02-16","arxiv_id":"2502.11157","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-cross-tokenizer-knowledge","slug":"enhancing-cross-tokenizer-knowledge","title":"Enhancing Cross-Tokenizer Knowledge Distillation with Contextual Dynamical Mapping","date":"2025-02-16","arxiv_id":"2502.11104","repositories_listed":1,"syntology":null},{"url":"/paper/mathematical-reasoning-in-large-language","slug":"mathematical-reasoning-in-large-language","title":"Mathematical Reasoning in Large Language Models: Assessing Logical and Arithmetic Errors across Wide Numerical Ranges","date":"2025-02-12","arxiv_id":"2502.08680","repositories_listed":1,"syntology":null},{"url":"/paper/codei-o-condensing-reasoning-patterns-via","slug":"codei-o-condensing-reasoning-patterns-via","title":"CodeI/O: Condensing Reasoning Patterns via Code Input-Output Prediction","date":"2025-02-11","arxiv_id":"2502.07316","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/codei-o-condensing-reasoning-patterns-via#ran","syntology_url":"https://syntology.ai/paper/2502.07316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07316"}},"official":{"repos":["hkust-nlp/codeio"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}}],"record_sha256":"937ac0d8a2cec7a6529323b3286833bc8841f54533e2371af3861ead36739fbf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}