{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/math/papers/ran/1","list_of":"/task/math","task":"Math","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":4,"rows_per_page":100,"rows":[1,100],"of":349,"counts":{"archive_papers_tagged":1596,"with_a_code_link":765,"where_syntology_ran_a_sample":349,"not_listed_spam_title":0,"listed":1596,"listed_where_code_ran":349,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/math/papers/ran/1","prev":null,"next":"/task/math/papers/ran/2","papers":[{"url":"/paper/questa-expanding-reasoning-capacity-in-llms","slug":"questa-expanding-reasoning-capacity-in-llms","title":"QuestA: Expanding Reasoning Capacity in LLMs via Question Augmentation","date":"2025-07-17","arxiv_id":"2507.13266","repositories_listed":0,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/questa-expanding-reasoning-capacity-in-llms#ran","syntology_url":"https://syntology.ai/paper/2507.13266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.13266"}},"official":null}},{"url":"/paper/reasoning-or-memorization-unreliable-results","slug":"reasoning-or-memorization-unreliable-results","title":"Reasoning or Memorization? Unreliable Results of Reinforcement Learning Due to Data Contamination","date":"2025-07-14","arxiv_id":"2507.10532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reasoning-or-memorization-unreliable-results#ran","syntology_url":"https://syntology.ai/paper/2507.10532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.10532"}},"official":{"repos":["wumingqi/LLM-Math-Evaluation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-practical-two-stage-recipe-for-mathematical","slug":"a-practical-two-stage-recipe-for-mathematical","title":"A Practical Two-Stage Recipe for Mathematical LLMs: Maximizing Accuracy with SFT and Efficiency with Reinforcement Learning","date":"2025-07-11","arxiv_id":"2507.08267","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-practical-two-stage-recipe-for-mathematical#ran","syntology_url":"https://syntology.ai/paper/2507.08267","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.08267"}},"official":{"repos":["analokmaus/kaggle-aimo2-fast-math-r1"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evoagentx-an-automated-framework-for-evolving","slug":"evoagentx-an-automated-framework-for-evolving","title":"EvoAgentX: An Automated Framework for Evolving Agentic Workflows","date":"2025-07-04","arxiv_id":"2507.03616","repositories_listed":1,"syntology":{"n":20,"n_ran":19,"n_constructed":0,"n_ran_checked":19,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":19,"n_pointer_only":16,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evoagentx-an-automated-framework-for-evolving#ran","syntology_url":"https://syntology.ai/paper/2507.03616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.03616"}},"official":{"repos":["evoagentx/evoagentx"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/energy-based-transformers-are-scalable","slug":"energy-based-transformers-are-scalable","title":"Energy-Based Transformers are Scalable Learners and Thinkers","date":"2025-07-02","arxiv_id":"2507.02092","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/energy-based-transformers-are-scalable#ran","syntology_url":"https://syntology.ai/paper/2507.02092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.02092"}},"official":{"repos":["alexiglad/EBT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/spiral-self-play-on-zero-sum-games","slug":"spiral-self-play-on-zero-sum-games","title":"SPIRAL: Self-Play on Zero-Sum Games Incentivizes Reasoning via Multi-Agent Multi-Turn Reinforcement Learning","date":"2025-06-30","arxiv_id":"2506.24119","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spiral-self-play-on-zero-sum-games#ran","syntology_url":"https://syntology.ai/paper/2506.24119","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.24119"}},"official":{"repos":["spiral-rl/spiral"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/octothinker-mid-training-incentivizes","slug":"octothinker-mid-training-incentivizes","title":"OctoThinker: Mid-training Incentivizes Reinforcement Learning Scaling","date":"2025-06-25","arxiv_id":"2506.20512","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":4,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/octothinker-mid-training-incentivizes#ran","syntology_url":"https://syntology.ai/paper/2506.20512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20512"}},"official":{"repos":["gair-nlp/octothinker"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reasonflux-prm-trajectory-aware-prms-for-long","slug":"reasonflux-prm-trajectory-aware-prms-for-long","title":"ReasonFlux-PRM: Trajectory-Aware PRMs for Long Chain-of-Thought Reasoning in LLMs","date":"2025-06-23","arxiv_id":"2506.18896","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reasonflux-prm-trajectory-aware-prms-for-long#ran","syntology_url":"https://syntology.ai/paper/2506.18896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.18896"}},"official":{"repos":["gen-verse/reasonflux"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/plan-for-speed-dilated-scheduling-for-masked","slug":"plan-for-speed-dilated-scheduling-for-masked","title":"Plan for Speed -- Dilated Scheduling for Masked Diffusion Language Models","date":"2025-06-23","arxiv_id":"2506.19037","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/plan-for-speed-dilated-scheduling-for-masked#ran","syntology_url":"https://syntology.ai/paper/2506.19037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.19037"}},"official":null}},{"url":"/paper/evolving-prompts-in-context-an-open-ended","slug":"evolving-prompts-in-context-an-open-ended","title":"Evolving Prompts In-Context: An Open-ended, Self-replicating Perspective","date":"2025-06-22","arxiv_id":"2506.17930","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evolving-prompts-in-context-an-open-ended#ran","syntology_url":"https://syntology.ai/paper/2506.17930","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.17930"}},"official":{"repos":["jianyu-cs/promptquine"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/steering-llm-thinking-with-budget-guidance","slug":"steering-llm-thinking-with-budget-guidance","title":"Steering LLM Thinking with Budget Guidance","date":"2025-06-16","arxiv_id":"2506.13752","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":11,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/steering-llm-thinking-with-budget-guidance#ran","syntology_url":"https://syntology.ai/paper/2506.13752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13752"}},"official":{"repos":["umass-embodied-agi/budgetguidance"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/treerl-llm-reinforcement-learning-with-on","slug":"treerl-llm-reinforcement-learning-with-on","title":"TreeRL: LLM Reinforcement Learning with On-Policy Tree Search","date":"2025-06-13","arxiv_id":"2506.11902","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/treerl-llm-reinforcement-learning-with-on#ran","syntology_url":"https://syntology.ai/paper/2506.11902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.11902"}},"official":{"repos":["thudm/treerl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/recut-balancing-reasoning-length-and-accuracy","slug":"recut-balancing-reasoning-length-and-accuracy","title":"ReCUT: Balancing Reasoning Length and Accuracy in LLMs via Stepwise Trails and Preference Optimization","date":"2025-06-12","arxiv_id":"2506.10822","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recut-balancing-reasoning-length-and-accuracy#ran","syntology_url":"https://syntology.ai/paper/2506.10822","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.10822"}},"official":{"repos":["neuir/recut"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sws-self-aware-weakness-driven-problem","slug":"sws-self-aware-weakness-driven-problem","title":"SwS: Self-aware Weakness-driven Problem Synthesis in Reinforcement Learning for LLM Reasoning","date":"2025-06-10","arxiv_id":"2506.08989","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sws-self-aware-weakness-driven-problem#ran","syntology_url":"https://syntology.ai/paper/2506.08989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08989"}},"official":{"repos":["mastervito/sws"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/play-to-generalize-learning-to-reason-through","slug":"play-to-generalize-learning-to-reason-through","title":"Play to Generalize: Learning to Reason Through Game Play","date":"2025-06-09","arxiv_id":"2506.08011","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/play-to-generalize-learning-to-reason-through#ran","syntology_url":"https://syntology.ai/paper/2506.08011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08011"}},"official":{"repos":["yunfeixie233/vigal"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/guided-speculative-inference-for-efficient","slug":"guided-speculative-inference-for-efficient","title":"Guided Speculative Inference for Efficient Test-Time Alignment of LLMs","date":"2025-06-04","arxiv_id":"2506.04118","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guided-speculative-inference-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2506.04118","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.04118"}},"official":{"repos":["j-geuter/gsi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/openthoughts-data-recipes-for-reasoning","slug":"openthoughts-data-recipes-for-reasoning","title":"OpenThoughts: Data Recipes for Reasoning Models","date":"2025-06-04","arxiv_id":"2506.04178","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/openthoughts-data-recipes-for-reasoning#ran","syntology_url":"https://syntology.ai/paper/2506.04178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.04178"}},"official":{"repos":["open-thoughts/open-thoughts"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/invariance-makes-llm-unlearning-resilient","slug":"invariance-makes-llm-unlearning-resilient","title":"Invariance Makes LLM Unlearning Resilient Even to Unanticipated Downstream Fine-Tuning","date":"2025-06-02","arxiv_id":"2506.01339","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/invariance-makes-llm-unlearning-resilient#ran","syntology_url":"https://syntology.ai/paper/2506.01339","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01339"}},"official":{"repos":["optml-group/unlearn-ilu"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-surprising-effectiveness-of-negative","slug":"the-surprising-effectiveness-of-negative","title":"The Surprising Effectiveness of Negative Reinforcement in LLM Reasoning","date":"2025-06-02","arxiv_id":"2506.01347","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-surprising-effectiveness-of-negative#ran","syntology_url":"https://syntology.ai/paper/2506.01347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01347"}},"official":{"repos":["tianhongzxy/rlvr-decomposed"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/areal-a-large-scale-asynchronous","slug":"areal-a-large-scale-asynchronous","title":"AReaL: A Large-Scale Asynchronous Reinforcement Learning System for Language Reasoning","date":"2025-05-30","arxiv_id":"2505.24298","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/areal-a-large-scale-asynchronous#ran","syntology_url":"https://syntology.ai/paper/2505.24298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24298"}},"official":{"repos":["inclusionai/areal"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-thought-efficient-reasoning-via","slug":"a-thought-efficient-reasoning-via","title":"A*-Thought: Efficient Reasoning via Bidirectional Compression for Low-Resource Settings","date":"2025-05-30","arxiv_id":"2505.24550","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-thought-efficient-reasoning-via#ran","syntology_url":"https://syntology.ai/paper/2505.24550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24550"}},"official":{"repos":["ai9stars/astar-thought"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/matharena-evaluating-llms-on-uncontaminated","slug":"matharena-evaluating-llms-on-uncontaminated","title":"MathArena: Evaluating LLMs on Uncontaminated Math Competitions","date":"2025-05-29","arxiv_id":"2505.23281","repositories_listed":1,"syntology":{"n":19,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/matharena-evaluating-llms-on-uncontaminated#ran","syntology_url":"https://syntology.ai/paper/2505.23281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23281"}},"official":{"repos":["eth-sri/matharena"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/discriminative-policy-optimization-for-token","slug":"discriminative-policy-optimization-for-token","title":"Discriminative Policy Optimization for Token-Level Reward Models","date":"2025-05-29","arxiv_id":"2505.23363","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/discriminative-policy-optimization-for-token#ran","syntology_url":"https://syntology.ai/paper/2505.23363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23363"}},"official":{"repos":["homzer/q-rm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-action-model-with-open-world","slug":"vision-language-action-model-with-open-world","title":"ChatVLA-2: Vision-Language-Action Model with Open-World Embodied Reasoning from Pretrained Knowledge","date":"2025-05-28","arxiv_id":"2505.21906","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vision-language-action-model-with-open-world#ran","syntology_url":"https://syntology.ai/paper/2505.21906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21906"}},"official":null}},{"url":"/paper/skywork-open-reasoner-1-technical-report","slug":"skywork-open-reasoner-1-technical-report","title":"Skywork Open Reasoner 1 Technical Report","date":"2025-05-28","arxiv_id":"2505.22312","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/skywork-open-reasoner-1-technical-report#ran","syntology_url":"https://syntology.ai/paper/2505.22312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22312"}},"official":{"repos":["skyworkai/skywork-or1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-post-training-for-multi-modal","slug":"unsupervised-post-training-for-multi-modal","title":"Unsupervised Post-Training for Multi-Modal LLM Reasoning via GRPO","date":"2025-05-28","arxiv_id":"2505.22453","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unsupervised-post-training-for-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2505.22453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22453"}},"official":{"repos":["hiyouga/easyr1","waltonfuture/mm-upt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcing-general-reasoning-without","slug":"reinforcing-general-reasoning-without","title":"Reinforcing General Reasoning without Verifiers","date":"2025-05-27","arxiv_id":"2505.21493","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcing-general-reasoning-without#ran","syntology_url":"https://syntology.ai/paper/2505.21493","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21493"}},"official":{"repos":["sail-sg/verifree"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/r2r-efficiently-navigating-divergent","slug":"r2r-efficiently-navigating-divergent","title":"R2R: Efficiently Navigating Divergent Reasoning Paths with Small-Large Model Token Routing","date":"2025-05-27","arxiv_id":"2505.21600","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r2r-efficiently-navigating-divergent#ran","syntology_url":"https://syntology.ai/paper/2505.21600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21600"}},"official":{"repos":["thu-nics/r2r"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-multimodal-large-language-model","slug":"unifying-multimodal-large-language-model","title":"Unifying Multimodal Large Language Model Capabilities and Modalities via Model Merging","date":"2025-05-26","arxiv_id":"2505.19892","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2505.19892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19892"}},"official":{"repos":["walkerworldpeace/mllmerging"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/does-representation-intervention-really","slug":"does-representation-intervention-really","title":"Does Representation Intervention Really Identify Desired Concepts and Elicit Alignment?","date":"2025-05-24","arxiv_id":"2505.18672","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/does-representation-intervention-really#ran","syntology_url":"https://syntology.ai/paper/2505.18672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18672"}},"official":null}},{"url":"/paper/value-guided-search-for-efficient-chain-of","slug":"value-guided-search-for-efficient-chain-of","title":"Value-Guided Search for Efficient Chain-of-Thought Reasoning","date":"2025-05-23","arxiv_id":"2505.17373","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/value-guided-search-for-efficient-chain-of#ran","syntology_url":"https://syntology.ai/paper/2505.17373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17373"}},"official":{"repos":["kaiwenw/value-guided-search"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/decoupled-visual-interpretation-and","slug":"decoupled-visual-interpretation-and","title":"Decoupled Visual Interpretation and Linguistic Reasoning for Math Problem Solving","date":"2025-05-23","arxiv_id":"2505.17609","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decoupled-visual-interpretation-and#ran","syntology_url":"https://syntology.ai/paper/2505.17609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17609"}},"official":{"repos":["guozix/dvlr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/saturn-sat-based-reinforcement-learning-to","slug":"saturn-sat-based-reinforcement-learning-to","title":"SATURN: SAT-based Reinforcement Learning to Unleash Language Model Reasoning","date":"2025-05-22","arxiv_id":"2505.16368","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/saturn-sat-based-reinforcement-learning-to#ran","syntology_url":"https://syntology.ai/paper/2505.16368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16368"}},"official":{"repos":["gtxygyzb/saturn-code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"url":"/paper/webagent-r1-training-web-agents-via-end-to","slug":"webagent-r1-training-web-agents-via-end-to","title":"WebAgent-R1: Training Web Agents via End-to-End Multi-Turn Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.16421","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/webagent-r1-training-web-agents-via-end-to#ran","syntology_url":"https://syntology.ai/paper/2505.16421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16421"}},"official":{"repos":["weizhepei/webagent-r1"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/conciserl-conciseness-guided-reinforcement","slug":"conciserl-conciseness-guided-reinforcement","title":"ConciseRL: Conciseness-Guided Reinforcement Learning for Efficient Reasoning Models","date":"2025-05-22","arxiv_id":"2505.17250","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conciserl-conciseness-guided-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2505.17250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17250"}},"official":{"repos":["razvandu/conciserl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rl-tango-reinforcing-generator-and-verifier","slug":"rl-tango-reinforcing-generator-and-verifier","title":"RL Tango: Reinforcing Generator and Verifier Together for Language Reasoning","date":"2025-05-21","arxiv_id":"2505.15034","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rl-tango-reinforcing-generator-and-verifier#ran","syntology_url":"https://syntology.ai/paper/2505.15034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15034"}},"official":{"repos":["kaiwenzha/rl-tango"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-should-we-enhance-the-safety-of-large","slug":"how-should-we-enhance-the-safety-of-large","title":"How Should We Enhance the Safety of Large Reasoning Models: An Empirical Study","date":"2025-05-21","arxiv_id":"2505.15404","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":6,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-should-we-enhance-the-safety-of-large#ran","syntology_url":"https://syntology.ai/paper/2505.15404","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15404"}},"official":{"repos":["thu-coai/lrm-safety-study"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tinyv-reducing-false-negatives-in","slug":"tinyv-reducing-false-negatives-in","title":"TinyV: Reducing False Negatives in Verification Improves RL for LLM Reasoning","date":"2025-05-20","arxiv_id":"2505.14625","repositories_listed":1,"syntology":{"n":19,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/tinyv-reducing-false-negatives-in#ran","syntology_url":"https://syntology.ai/paper/2505.14625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14625"}},"official":{"repos":["uw-nsl/tinyv"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/seek-in-the-dark-reasoning-via-test-time","slug":"seek-in-the-dark-reasoning-via-test-time","title":"Seek in the Dark: Reasoning via Test-Time Instance-Level Policy Gradient in Latent Space","date":"2025-05-19","arxiv_id":"2505.13308","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/seek-in-the-dark-reasoning-via-test-time#ran","syntology_url":"https://syntology.ai/paper/2505.13308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13308"}},"official":{"repos":["bigai-nlco/latentseek"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/thinkless-llm-learns-when-to-think","slug":"thinkless-llm-learns-when-to-think","title":"Thinkless: LLM Learns When to Think","date":"2025-05-19","arxiv_id":"2505.13379","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/thinkless-llm-learns-when-to-think#ran","syntology_url":"https://syntology.ai/paper/2505.13379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13379"}},"official":{"repos":["vainf/thinkless"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptthink-reasoning-models-can-learn-when-to","slug":"adaptthink-reasoning-models-can-learn-when-to","title":"AdaptThink: Reasoning Models Can Learn When to Think","date":"2025-05-19","arxiv_id":"2505.13417","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":8,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 2 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adaptthink-reasoning-models-can-learn-when-to#ran","syntology_url":"https://syntology.ai/paper/2505.13417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13417"}},"official":{"repos":["thu-keg/adaptthink"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-prm-enhancing-multimodal-mathematical","slug":"mm-prm-enhancing-multimodal-mathematical","title":"MM-PRM: Enhancing Multimodal Mathematical Reasoning with Scalable Step-Level Supervision","date":"2025-05-19","arxiv_id":"2505.13427","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mm-prm-enhancing-multimodal-mathematical#ran","syntology_url":"https://syntology.ai/paper/2505.13427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13427"}},"official":{"repos":["modalminds/mm-prm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/marge-improving-math-reasoning-for-llms-with","slug":"marge-improving-math-reasoning-for-llms-with","title":"MARGE: Improving Math Reasoning for LLMs with Guided Exploration","date":"2025-05-18","arxiv_id":"2505.12500","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/marge-improving-math-reasoning-for-llms-with#ran","syntology_url":"https://syntology.ai/paper/2505.12500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12500"}},"official":{"repos":["georgao35/marge"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","unlocated"]}}},{"url":"/paper/synthetic-data-rl-task-definition-is-all-you","slug":"synthetic-data-rl-task-definition-is-all-you","title":"Synthetic Data RL: Task Definition Is All You Need","date":"2025-05-18","arxiv_id":"2505.17063","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/synthetic-data-rl-task-definition-is-all-you#ran","syntology_url":"https://syntology.ai/paper/2505.17063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17063"}},"official":{"repos":["gydpku/data_synthesis_rl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mathcoder-vl-bridging-vision-and-code-for","slug":"mathcoder-vl-bridging-vision-and-code-for","title":"MathCoder-VL: Bridging Vision and Code for Enhanced Multimodal Mathematical Reasoning","date":"2025-05-15","arxiv_id":"2505.10557","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathcoder-vl-bridging-vision-and-code-for#ran","syntology_url":"https://syntology.ai/paper/2505.10557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10557"}},"official":{"repos":["mathllm/mathcoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/agent-rl-scaling-law-agent-rl-with","slug":"agent-rl-scaling-law-agent-rl-with","title":"Agent RL Scaling Law: Agent RL with Spontaneous Code Execution for Mathematical Problem Solving","date":"2025-05-12","arxiv_id":"2505.07773","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agent-rl-scaling-law-agent-rl-with#ran","syntology_url":"https://syntology.ai/paper/2505.07773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07773"}},"official":{"repos":["anonymize-author/agentrl","yyht/openrlhf_async_pipline"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rm-r1-reward-modeling-as-reasoning","slug":"rm-r1-reward-modeling-as-reasoning","title":"RM-R1: Reward Modeling as Reasoning","date":"2025-05-05","arxiv_id":"2505.02387","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/rm-r1-reward-modeling-as-reasoning#ran","syntology_url":"https://syntology.ai/paper/2505.02387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02387"}},"official":{"repos":["rm-r1-uiuc/rm-r1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rewriting-pre-training-data-boosts-llm","slug":"rewriting-pre-training-data-boosts-llm","title":"Rewriting Pre-Training Data Boosts LLM Performance in Math and Code","date":"2025-05-05","arxiv_id":"2505.02881","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rewriting-pre-training-data-boosts-llm#ran","syntology_url":"https://syntology.ai/paper/2505.02881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02881"}},"official":{"repos":["rioyokotalab/swallow-code-math"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepcritic-deliberate-critique-with-large","slug":"deepcritic-deliberate-critique-with-large","title":"DeepCritic: Deliberate Critique with Large Language Models","date":"2025-05-01","arxiv_id":"2505.00662","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepcritic-deliberate-critique-with-large#ran","syntology_url":"https://syntology.ai/paper/2505.00662","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.00662"}},"official":{"repos":["rucbm/deepcritic"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-reasoning-in-large","slug":"reinforcement-learning-for-reasoning-in-large","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","date":"2025-04-29","arxiv_id":"2504.20571","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":9,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/reinforcement-learning-for-reasoning-in-large#ran","syntology_url":"https://syntology.ai/paper/2504.20571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20571"}},"official":{"repos":["ypwang61/one-shot-rlvr"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-reasoning-for-llms-through","slug":"efficient-reasoning-for-llms-through","title":"Efficient Reasoning for LLMs through Speculative Chain-of-Thought","date":"2025-04-27","arxiv_id":"2504.19095","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-reasoning-for-llms-through#ran","syntology_url":"https://syntology.ai/paper/2504.19095","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.19095"}},"official":{"repos":["jikai0wang/speculative_cot"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-on-prompt-compression-for","slug":"an-empirical-study-on-prompt-compression-for","title":"An Empirical Study on Prompt Compression for Large Language Models","date":"2025-04-24","arxiv_id":"2505.00019","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-on-prompt-compression-for#ran","syntology_url":"https://syntology.ai/paper/2505.00019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.00019"}},"official":{"repos":["3DAgentWorld/Toolkit-for-Prompt-Compression"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/process-reward-models-that-think","slug":"process-reward-models-that-think","title":"Process Reward Models That Think","date":"2025-04-23","arxiv_id":"2504.16828","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/process-reward-models-that-think#ran","syntology_url":"https://syntology.ai/paper/2504.16828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16828"}},"official":{"repos":["mukhal/thinkprm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/aimo-2-winning-solution-building-state-of-the","slug":"aimo-2-winning-solution-building-state-of-the","title":"AIMO-2 Winning Solution: Building State-of-the-Art Mathematical Reasoning Models with OpenMathReasoning dataset","date":"2025-04-23","arxiv_id":"2504.16891","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/aimo-2-winning-solution-building-state-of-the#ran","syntology_url":"https://syntology.ai/paper/2504.16891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16891"}},"official":null}},{"url":"/paper/dianjin-r1-evaluating-and-enhancing-financial","slug":"dianjin-r1-evaluating-and-enhancing-financial","title":"DianJin-R1: Evaluating and Enhancing Financial Reasoning in Large Language Models","date":"2025-04-22","arxiv_id":"2504.15716","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dianjin-r1-evaluating-and-enhancing-financial#ran","syntology_url":"https://syntology.ai/paper/2504.15716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15716"}},"official":{"repos":["aliyun/qwen-dianjin"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-early-exit-in-reasoning-models","slug":"dynamic-early-exit-in-reasoning-models","title":"Dynamic Early Exit in Reasoning Models","date":"2025-04-22","arxiv_id":"2504.15895","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-early-exit-in-reasoning-models#ran","syntology_url":"https://syntology.ai/paper/2504.15895","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15895"}},"official":{"repos":["iie-ycx/deer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ttrl-test-time-reinforcement-learning","slug":"ttrl-test-time-reinforcement-learning","title":"TTRL: Test-Time Reinforcement Learning","date":"2025-04-22","arxiv_id":"2504.16084","repositories_listed":3,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":6,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ttrl-test-time-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2504.16084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16084"}},"official":{"repos":["prime-rl/ttrl","tsinghuac3i/awesome-rl-reasoning-recipes"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-judges-as-evaluators-the-jetts","slug":"evaluating-judges-as-evaluators-the-jetts","title":"Evaluating Judges as Evaluators: The JETTS Benchmark of LLM-as-Judges as Test-Time Scaling Evaluators","date":"2025-04-21","arxiv_id":"2504.15253","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-judges-as-evaluators-the-jetts#ran","syntology_url":"https://syntology.ai/paper/2504.15253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15253"}},"official":{"repos":["salesforceairesearch/jetts-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stop-summation-min-form-credit-assignment-is","slug":"stop-summation-min-form-credit-assignment-is","title":"Stop Summation: Min-Form Credit Assignment Is All Process Reward Model Needs for Reasoning","date":"2025-04-21","arxiv_id":"2504.15275","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stop-summation-min-form-credit-assignment-is#ran","syntology_url":"https://syntology.ai/paper/2504.15275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15275"}},"official":{"repos":["cjreinforce/pure"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-process-reward-model-training-via","slug":"efficient-process-reward-model-training-via","title":"Efficient Process Reward Model Training via Active Learning","date":"2025-04-14","arxiv_id":"2504.10559","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-process-reward-model-training-via#ran","syntology_url":"https://syntology.ai/paper/2504.10559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10559"}},"official":{"repos":["sail-sg/activeprm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-jailbreak-tax-how-useful-are-your","slug":"the-jailbreak-tax-how-useful-are-your","title":"The Jailbreak Tax: How Useful are Your Jailbreak Outputs?","date":"2025-04-14","arxiv_id":"2504.10694","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-jailbreak-tax-how-useful-are-your#ran","syntology_url":"https://syntology.ai/paper/2504.10694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10694"}},"official":{"repos":["ethz-spylab/jailbreak-tax"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-cheatsheet-test-time-learning-with","slug":"dynamic-cheatsheet-test-time-learning-with","title":"Dynamic Cheatsheet: Test-Time Learning with Adaptive Memory","date":"2025-04-10","arxiv_id":"2504.07952","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-cheatsheet-test-time-learning-with#ran","syntology_url":"https://syntology.ai/paper/2504.07952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07952"}},"official":{"repos":["suzgunmirac/dynamic-cheatsheet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-rethinker-incentivizing-self-reflection-of","slug":"vl-rethinker-incentivizing-self-reflection-of","title":"VL-Rethinker: Incentivizing Self-Reflection of Vision-Language Models with Reinforcement Learning","date":"2025-04-10","arxiv_id":"2504.08837","repositories_listed":2,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vl-rethinker-incentivizing-self-reflection-of#ran","syntology_url":"https://syntology.ai/paper/2504.08837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.08837"}},"official":null}},{"url":"/paper/right-question-is-already-half-the-answer","slug":"right-question-is-already-half-the-answer","title":"Right Question is Already Half the Answer: Fully Unsupervised LLM Reasoning Incentivization","date":"2025-04-08","arxiv_id":"2504.05812","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/right-question-is-already-half-the-answer#ran","syntology_url":"https://syntology.ai/paper/2504.05812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05812"}},"official":{"repos":["qingyangzhang/empo"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/efficient-reinforcement-finetuning-via","slug":"efficient-reinforcement-finetuning-via","title":"Efficient Reinforcement Finetuning via Adaptive Curriculum Learning","date":"2025-04-07","arxiv_id":"2504.05520","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/efficient-reinforcement-finetuning-via#ran","syntology_url":"https://syntology.ai/paper/2504.05520","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05520"}},"official":{"repos":["uscnlp-lime/verl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/large-vision-language-models-are-unsupervised","slug":"large-vision-language-models-are-unsupervised","title":"Large (Vision) Language Models are Unsupervised In-Context Learners","date":"2025-04-03","arxiv_id":"2504.02349","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/large-vision-language-models-are-unsupervised#ran","syntology_url":"https://syntology.ai/paper/2504.02349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02349"}},"official":{"repos":["mlbio-epfl/joint-inference"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/megamath-pushing-the-limits-of-open-math","slug":"megamath-pushing-the-limits-of-open-math","title":"MegaMath: Pushing the Limits of Open Math Corpora","date":"2025-04-03","arxiv_id":"2504.02807","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/megamath-pushing-the-limits-of-open-math#ran","syntology_url":"https://syntology.ai/paper/2504.02807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02807"}},"official":{"repos":["llm360/megamath"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/entropy-based-adaptive-weighting-for-self","slug":"entropy-based-adaptive-weighting-for-self","title":"Entropy-Based Adaptive Weighting for Self-Training","date":"2025-03-31","arxiv_id":"2503.23913","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/entropy-based-adaptive-weighting-for-self#ran","syntology_url":"https://syntology.ai/paper/2503.23913","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23913"}},"official":{"repos":["mandyyyyii/east"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inference-time-scaling-for-complex-tasks","slug":"inference-time-scaling-for-complex-tasks","title":"Inference-Time Scaling for Complex Tasks: Where We Stand and What Lies Ahead","date":"2025-03-31","arxiv_id":"2504.00294","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inference-time-scaling-for-complex-tasks#ran","syntology_url":"https://syntology.ai/paper/2504.00294","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.00294"}},"official":null}},{"url":"/paper/cppo-accelerating-the-training-of-group","slug":"cppo-accelerating-the-training-of-group","title":"CPPO: Accelerating the Training of Group Relative Policy Optimization-Based Reasoning Models","date":"2025-03-28","arxiv_id":"2503.22342","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/cppo-accelerating-the-training-of-group#ran","syntology_url":"https://syntology.ai/paper/2503.22342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.22342"}},"official":{"repos":["lzhxmu/cppo"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/questbench-can-llms-ask-the-right-question-to","slug":"questbench-can-llms-ask-the-right-question-to","title":"QuestBench: Can LLMs ask the right question to acquire information in reasoning tasks?","date":"2025-03-28","arxiv_id":"2503.22674","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/questbench-can-llms-ask-the-right-question-to#ran","syntology_url":"https://syntology.ai/paper/2503.22674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.22674"}},"official":{"repos":["google-deepmind/questbench"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-reason-for-long-form-story","slug":"learning-to-reason-for-long-form-story","title":"Learning to Reason for Long-Form Story Generation","date":"2025-03-28","arxiv_id":"2503.22828","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":1,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-reason-for-long-form-story#ran","syntology_url":"https://syntology.ai/paper/2503.22828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.22828"}},"official":{"repos":["Alex-Gurung/ReasoningNCP"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/challenging-the-boundaries-of-reasoning-an","slug":"challenging-the-boundaries-of-reasoning-an","title":"Challenging the Boundaries of Reasoning: An Olympiad-Level Math Benchmark for Large Language Models","date":"2025-03-27","arxiv_id":"2503.21380","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/challenging-the-boundaries-of-reasoning-an#ran","syntology_url":"https://syntology.ai/paper/2503.21380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21380"}},"official":{"repos":["RUCAIBox/Slow_Thinking_with_LLMs","rucaibox/olymmath"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/effective-skill-unlearning-through","slug":"effective-skill-unlearning-through","title":"Effective Skill Unlearning through Intervention and Abstention","date":"2025-03-27","arxiv_id":"2503.21730","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/effective-skill-unlearning-through#ran","syntology_url":"https://syntology.ai/paper/2503.21730","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21730"}},"official":{"repos":["trustworthy-ml-lab/effective_skill_unlearning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reasoning-to-learn-from-latent-thoughts","slug":"reasoning-to-learn-from-latent-thoughts","title":"Reasoning to Learn from Latent Thoughts","date":"2025-03-24","arxiv_id":"2503.18866","repositories_listed":1,"syntology":{"n":20,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":4,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/reasoning-to-learn-from-latent-thoughts#ran","syntology_url":"https://syntology.ai/paper/2503.18866","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18866"}},"official":null}},{"url":"/paper/simplerl-zoo-investigating-and-taming-zero","slug":"simplerl-zoo-investigating-and-taming-zero","title":"SimpleRL-Zoo: Investigating and Taming Zero Reinforcement Learning for Open Base Models in the Wild","date":"2025-03-24","arxiv_id":"2503.18892","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simplerl-zoo-investigating-and-taming-zero#ran","syntology_url":"https://syntology.ai/paper/2503.18892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18892"}},"official":null}},{"url":"/paper/agentrxiv-towards-collaborative-autonomous","slug":"agentrxiv-towards-collaborative-autonomous","title":"AgentRxiv: Towards Collaborative Autonomous Research","date":"2025-03-23","arxiv_id":"2503.18102","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/agentrxiv-towards-collaborative-autonomous#ran","syntology_url":"https://syntology.ai/paper/2503.18102","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18102"}},"official":null}},{"url":"/paper/chatbench-from-static-benchmarks-to-human-ai","slug":"chatbench-from-static-benchmarks-to-human-ai","title":"ChatBench: From Static Benchmarks to Human-AI Evaluation","date":"2025-03-22","arxiv_id":"2504.07114","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chatbench-from-static-benchmarks-to-human-ai#ran","syntology_url":"https://syntology.ai/paper/2504.07114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07114"}},"official":{"repos":["serinachang5/interactive-eval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/xlstm-7b-a-recurrent-llm-for-fast-and","slug":"xlstm-7b-a-recurrent-llm-for-fast-and","title":"xLSTM 7B: A Recurrent LLM for Fast and Efficient Inference","date":"2025-03-17","arxiv_id":"2503.13427","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/xlstm-7b-a-recurrent-llm-for-fast-and#ran","syntology_url":"https://syntology.ai/paper/2503.13427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13427"}},"official":{"repos":["nx-ai/xlstm","nx-ai/xlstm-jax","nx-ai/mlstm_kernels"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exaone-deep-reasoning-enhanced-language","slug":"exaone-deep-reasoning-enhanced-language","title":"EXAONE Deep: Reasoning Enhanced Language Models","date":"2025-03-16","arxiv_id":"2503.12524","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exaone-deep-reasoning-enhanced-language#ran","syntology_url":"https://syntology.ai/paper/2503.12524","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.12524"}},"official":null}},{"url":"/paper/light-r1-curriculum-sft-dpo-and-rl-for-long","slug":"light-r1-curriculum-sft-dpo-and-rl-for-long","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","date":"2025-03-13","arxiv_id":"2503.10460","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/light-r1-curriculum-sft-dpo-and-rl-for-long#ran","syntology_url":"https://syntology.ai/paper/2503.10460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10460"}},"official":{"repos":["qihoo360/light-r1"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/visualwebinstruct-scaling-up-multimodal","slug":"visualwebinstruct-scaling-up-multimodal","title":"VisualWebInstruct: Scaling up Multimodal Instruction Data through Web Search","date":"2025-03-13","arxiv_id":"2503.10582","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/visualwebinstruct-scaling-up-multimodal#ran","syntology_url":"https://syntology.ai/paper/2503.10582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10582"}},"official":null}},{"url":"/paper/promptcot-synthesizing-olympiad-level","slug":"promptcot-synthesizing-olympiad-level","title":"PromptCoT: Synthesizing Olympiad-level Problems for Mathematical Reasoning in Large Language Models","date":"2025-03-04","arxiv_id":"2503.02324","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/promptcot-synthesizing-olympiad-level#ran","syntology_url":"https://syntology.ai/paper/2503.02324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.02324"}},"official":{"repos":["zhaoxlpku/promptcot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-training-elicits-concise-reasoning-in","slug":"self-training-elicits-concise-reasoning-in","title":"Self-Training Elicits Concise Reasoning in Large Language Models","date":"2025-02-27","arxiv_id":"2502.20122","repositories_listed":1,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-training-elicits-concise-reasoning-in#ran","syntology_url":"https://syntology.ai/paper/2502.20122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20122"}},"official":{"repos":["tergelmunkhbat/concise-reasoning"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/finereason-evaluating-and-improving-llms","slug":"finereason-evaluating-and-improving-llms","title":"FINEREASON: Evaluating and Improving LLMs' Deliberate Reasoning through Reflective Puzzle Solving","date":"2025-02-27","arxiv_id":"2502.20238","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/finereason-evaluating-and-improving-llms#ran","syntology_url":"https://syntology.ai/paper/2502.20238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20238"}},"official":{"repos":["DAMO-NLP-SG/FineReason"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/big-math-a-large-scale-high-quality-math","slug":"big-math-a-large-scale-high-quality-math","title":"Big-Math: A Large-Scale, High-Quality Math Dataset for Reinforcement Learning in Language Models","date":"2025-02-24","arxiv_id":"2502.17387","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/big-math-a-large-scale-high-quality-math#ran","syntology_url":"https://syntology.ai/paper/2502.17387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17387"}},"official":{"repos":["synthlabsai/big-math"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/linguistic-generalizability-of-test-time","slug":"linguistic-generalizability-of-test-time","title":"Linguistic Generalizability of Test-Time Scaling in Mathematical Reasoning","date":"2025-02-24","arxiv_id":"2502.17407","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/linguistic-generalizability-of-test-time#ran","syntology_url":"https://syntology.ai/paper/2502.17407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17407"}},"official":{"repos":["gauss5930/mclm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/forgotten-polygons-multimodal-large-language","slug":"forgotten-polygons-multimodal-large-language","title":"Forgotten Polygons: Multimodal Large Language Models are Shape-Blind","date":"2025-02-21","arxiv_id":"2502.15969","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/forgotten-polygons-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2502.15969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15969"}},"official":{"repos":["rsinghlab/shape-blind"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/earlier-tokens-contribute-more-learning","slug":"earlier-tokens-contribute-more-learning","title":"Earlier Tokens Contribute More: Learning Direct Preference Optimization From Temporal Decay Perspective","date":"2025-02-20","arxiv_id":"2502.14340","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/earlier-tokens-contribute-more-learning#ran","syntology_url":"https://syntology.ai/paper/2502.14340","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14340"}},"official":{"repos":["lotusrc/d2po"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/s-test-time-scaling-for-code-generation","slug":"s-test-time-scaling-for-code-generation","title":"S*: Test Time Scaling for Code Generation","date":"2025-02-20","arxiv_id":"2502.14382","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/s-test-time-scaling-for-code-generation#ran","syntology_url":"https://syntology.ai/paper/2502.14382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14382"}},"official":{"repos":["novasky-ai/skythought"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cer-confidence-enhanced-reasoning-in-llms","slug":"cer-confidence-enhanced-reasoning-in-llms","title":"CER: Confidence Enhanced Reasoning in LLMs","date":"2025-02-20","arxiv_id":"2502.14634","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":0,"n_instrument":9,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":15,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 9 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/cer-confidence-enhanced-reasoning-in-llms#ran","syntology_url":"https://syntology.ai/paper/2502.14634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14634"}},"official":{"repos":["sharif-ml-lab/CER"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/logic-rl-unleashing-llm-reasoning-with-rule","slug":"logic-rl-unleashing-llm-reasoning-with-rule","title":"Logic-RL: Unleashing LLM Reasoning with Rule-Based Reinforcement Learning","date":"2025-02-20","arxiv_id":"2502.14768","repositories_listed":4,"syntology":{"n":23,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/logic-rl-unleashing-llm-reasoning-with-rule#ran","syntology_url":"https://syntology.ai/paper/2502.14768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14768"}},"official":{"repos":["Unakar/Logic-RL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/treecut-a-synthetic-unanswerable-math-word","slug":"treecut-a-synthetic-unanswerable-math-word","title":"TreeCut: A Synthetic Unanswerable Math Word Problem Dataset for LLM Hallucination Evaluation","date":"2025-02-19","arxiv_id":"2502.13442","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/treecut-a-synthetic-unanswerable-math-word#ran","syntology_url":"https://syntology.ai/paper/2502.13442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13442"}},"official":{"repos":["j-bagel/treecut-math"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sift-grounding-llm-reasoning-in-contexts-via","slug":"sift-grounding-llm-reasoning-in-contexts-via","title":"SIFT: Grounding LLM Reasoning in Contexts via Stickers","date":"2025-02-19","arxiv_id":"2502.14922","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sift-grounding-llm-reasoning-in-contexts-via#ran","syntology_url":"https://syntology.ai/paper/2502.14922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14922"}},"official":{"repos":["zhijie-group/sift"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/s-2-r-teaching-llms-to-self-verify-and-self","slug":"s-2-r-teaching-llms-to-self-verify-and-self","title":"S$^2$R: Teaching LLMs to Self-verify and Self-correct via Reinforcement Learning","date":"2025-02-18","arxiv_id":"2502.12853","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/s-2-r-teaching-llms-to-self-verify-and-self#ran","syntology_url":"https://syntology.ai/paper/2502.12853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.12853"}},"official":{"repos":["nineabyss/s2r"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/don-t-get-lost-in-the-trees-streamlining-llm","slug":"don-t-get-lost-in-the-trees-streamlining-llm","title":"Don't Get Lost in the Trees: Streamlining LLM Reasoning by Overcoming Tree Search Exploration Pitfalls","date":"2025-02-16","arxiv_id":"2502.11183","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/don-t-get-lost-in-the-trees-streamlining-llm#ran","syntology_url":"https://syntology.ai/paper/2502.11183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11183"}},"official":{"repos":["soistesimmer/fetch"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/rethinking-fine-tuning-when-scaling-test-time","slug":"rethinking-fine-tuning-when-scaling-test-time","title":"Rethinking Fine-Tuning when Scaling Test-Time Compute: Limiting Confidence Improves Mathematical Reasoning","date":"2025-02-11","arxiv_id":"2502.07154","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rethinking-fine-tuning-when-scaling-test-time#ran","syntology_url":"https://syntology.ai/paper/2502.07154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07154"}},"official":{"repos":["allanraventos/refine"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/goedel-prover-a-frontier-model-for-open","slug":"goedel-prover-a-frontier-model-for-open","title":"Goedel-Prover: A Frontier Model for Open-Source Automated Theorem Proving","date":"2025-02-11","arxiv_id":"2502.07640","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/goedel-prover-a-frontier-model-for-open#ran","syntology_url":"https://syntology.ai/paper/2502.07640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07640"}},"official":{"repos":["Goedel-LM/Goedel-Prover"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-1b-llm-surpass-405b-llm-rethinking","slug":"can-1b-llm-surpass-405b-llm-rethinking","title":"Can 1B LLM Surpass 405B LLM? Rethinking Compute-Optimal Test-Time Scaling","date":"2025-02-10","arxiv_id":"2502.06703","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":7,"n_pointer_only":4,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/can-1b-llm-surpass-405b-llm-rethinking#ran","syntology_url":"https://syntology.ai/paper/2502.06703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06703"}},"official":{"repos":["RyanLiu112/compute-optimal-tts"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-emergence-of-thinking-in-llms-i","slug":"on-the-emergence-of-thinking-in-llms-i","title":"On the Emergence of Thinking in LLMs I: Searching for the Right Intuition","date":"2025-02-10","arxiv_id":"2502.06773","repositories_listed":4,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":9,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-the-emergence-of-thinking-in-llms-i#ran","syntology_url":"https://syntology.ai/paper/2502.06773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06773"}},"official":{"repos":["GuanghaoYe/Emergence-of-Thinking"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"ee6e8bfe9cd978a9da5cb1391a0746bdc0339870e185f2b65200fa9f03906504","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}