{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/ran/1","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":15,"rows_per_page":100,"rows":[1,100],"of":1416,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1/papers/ran/1","prev":null,"next":"/task/reinforcement-learning-1/papers/ran/2","papers":[{"url":"/paper/questa-expanding-reasoning-capacity-in-llms","slug":"questa-expanding-reasoning-capacity-in-llms","title":"QuestA: Expanding Reasoning Capacity in LLMs via Question Augmentation","date":"2025-07-17","arxiv_id":"2507.13266","repositories_listed":0,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/questa-expanding-reasoning-capacity-in-llms#ran","syntology_url":"https://syntology.ai/paper/2507.13266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.13266"}},"official":null}},{"url":"/paper/reasoning-or-memorization-unreliable-results","slug":"reasoning-or-memorization-unreliable-results","title":"Reasoning or Memorization? Unreliable Results of Reinforcement Learning Due to Data Contamination","date":"2025-07-14","arxiv_id":"2507.10532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reasoning-or-memorization-unreliable-results#ran","syntology_url":"https://syntology.ai/paper/2507.10532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.10532"}},"official":{"repos":["wumingqi/LLM-Math-Evaluation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-practical-two-stage-recipe-for-mathematical","slug":"a-practical-two-stage-recipe-for-mathematical","title":"A Practical Two-Stage Recipe for Mathematical LLMs: Maximizing Accuracy with SFT and Efficiency with Reinforcement Learning","date":"2025-07-11","arxiv_id":"2507.08267","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-practical-two-stage-recipe-for-mathematical#ran","syntology_url":"https://syntology.ai/paper/2507.08267","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.08267"}},"official":{"repos":["analokmaus/kaggle-aimo2-fast-math-r1"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-rl-to-long-videos","slug":"scaling-rl-to-long-videos","title":"Scaling RL to Long Videos","date":"2025-07-10","arxiv_id":"2507.07966","repositories_listed":1,"syntology":{"n":24,"n_ran":18,"n_constructed":1,"n_ran_checked":14,"n_instrument":4,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":13,"n_pointer_only":1,"phrase":"18 ran (of which 1 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 1 violated, 13 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/scaling-rl-to-long-videos#ran","syntology_url":"https://syntology.ai/paper/2507.07966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.07966"}},"official":{"repos":["hiyouga/easyr1"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/gta1-gui-test-time-scaling-agent","slug":"gta1-gui-test-time-scaling-agent","title":"GTA1: GUI Test-time Scaling Agent","date":"2025-07-08","arxiv_id":"2507.05791","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gta1-gui-test-time-scaling-agent#ran","syntology_url":"https://syntology.ai/paper/2507.05791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.05791"}},"official":{"repos":["yan98/gta1"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kwai-keye-vl-technical-report","slug":"kwai-keye-vl-technical-report","title":"Kwai Keye-VL Technical Report","date":"2025-07-02","arxiv_id":"2507.01949","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 3 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kwai-keye-vl-technical-report#ran","syntology_url":"https://syntology.ai/paper/2507.01949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.01949"}},"official":{"repos":["kwai-keye/keye"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rag-r1-incentivize-the-search-and-reasoning","slug":"rag-r1-incentivize-the-search-and-reasoning","title":"RAG-R1 : Incentivize the Search and Reasoning Capabilities of LLMs through Multi-query Parallelism","date":"2025-06-30","arxiv_id":"2507.02962","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rag-r1-incentivize-the-search-and-reasoning#ran","syntology_url":"https://syntology.ai/paper/2507.02962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.02962"}},"official":null}},{"url":"/paper/seg-r1-segmentation-can-be-surprisingly","slug":"seg-r1-segmentation-can-be-surprisingly","title":"Seg-R1: Segmentation Can Be Surprisingly Simple with Reinforcement Learning","date":"2025-06-27","arxiv_id":"2506.22624","repositories_listed":2,"syntology":{"n":23,"n_ran":22,"n_constructed":0,"n_ran_checked":13,"n_instrument":9,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":4,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 9 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/seg-r1-segmentation-can-be-surprisingly#ran","syntology_url":"https://syntology.ai/paper/2506.22624","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.22624"}},"official":{"repos":["geshang777/Seg-R1"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/flow-based-single-step-completion-for","slug":"flow-based-single-step-completion-for","title":"Flow-Based Single-Step Completion for Efficient and Expressive Policy Learning","date":"2025-06-26","arxiv_id":"2506.21427","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flow-based-single-step-completion-for#ran","syntology_url":"https://syntology.ai/paper/2506.21427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.21427"}},"official":null}},{"url":"/paper/octothinker-mid-training-incentivizes","slug":"octothinker-mid-training-incentivizes","title":"OctoThinker: Mid-training Incentivizes Reinforcement Learning Scaling","date":"2025-06-25","arxiv_id":"2506.20512","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":4,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/octothinker-mid-training-incentivizes#ran","syntology_url":"https://syntology.ai/paper/2506.20512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20512"}},"official":{"repos":["gair-nlp/octothinker"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-increases-wind-farm","slug":"reinforcement-learning-increases-wind-farm","title":"Reinforcement Learning Increases Wind Farm Power Production by Enabling Closed-Loop Collaborative Control","date":"2025-06-25","arxiv_id":"2506.20554","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-increases-wind-farm#ran","syntology_url":"https://syntology.ai/paper/2506.20554","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20554"}},"official":{"repos":["admole/wind-rl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diffucoder-understanding-and-improving-masked","slug":"diffucoder-understanding-and-improving-masked","title":"DiffuCoder: Understanding and Improving Masked Diffusion Models for Code Generation","date":"2025-06-25","arxiv_id":"2506.20639","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diffucoder-understanding-and-improving-masked#ran","syntology_url":"https://syntology.ai/paper/2506.20639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20639"}},"official":{"repos":["apple/ml-diffucoder"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/longwriter-zero-mastering-ultra-long-text","slug":"longwriter-zero-mastering-ultra-long-text","title":"LongWriter-Zero: Mastering Ultra-Long Text Generation via Reinforcement Learning","date":"2025-06-23","arxiv_id":"2506.18841","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":6,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/longwriter-zero-mastering-ultra-long-text#ran","syntology_url":"https://syntology.ai/paper/2506.18841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.18841"}},"official":null}},{"url":"/paper/minimax-m1-scaling-test-time-compute","slug":"minimax-m1-scaling-test-time-compute","title":"MiniMax-M1: Scaling Test-Time Compute Efficiently with Lightning Attention","date":"2025-06-16","arxiv_id":"2506.13585","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minimax-m1-scaling-test-time-compute#ran","syntology_url":"https://syntology.ai/paper/2506.13585","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13585"}},"official":{"repos":["minimax-ai/minimax-m1"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-courage-to-stop-overcoming-sunk-cost","slug":"the-courage-to-stop-overcoming-sunk-cost","title":"The Courage to Stop: Overcoming Sunk Cost Fallacy in Deep Reinforcement Learning","date":"2025-06-16","arxiv_id":"2506.13672","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-courage-to-stop-overcoming-sunk-cost#ran","syntology_url":"https://syntology.ai/paper/2506.13672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13672"}},"official":null}},{"url":"/paper/enhancing-rating-based-reinforcement-learning","slug":"enhancing-rating-based-reinforcement-learning","title":"Enhancing Rating-Based Reinforcement Learning to Effectively Leverage Feedback from Large Vision-Language Models","date":"2025-06-15","arxiv_id":"2506.12822","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-rating-based-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2506.12822","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12822"}},"official":{"repos":["tunglm2203/erlvlm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/soundmind-rl-incentivized-logic-reasoning-for","slug":"soundmind-rl-incentivized-logic-reasoning-for","title":"SoundMind: RL-Incentivized Logic Reasoning for Audio-Language Models","date":"2025-06-15","arxiv_id":"2506.12935","repositories_listed":1,"syntology":{"n":16,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/soundmind-rl-incentivized-logic-reasoning-for#ran","syntology_url":"https://syntology.ai/paper/2506.12935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12935"}},"official":{"repos":["xid32/soundmind"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/dr-sac-distributionally-robust-soft-actor","slug":"dr-sac-distributionally-robust-soft-actor","title":"DR-SAC: Distributionally Robust Soft Actor-Critic for Reinforcement Learning under Uncertainty","date":"2025-06-14","arxiv_id":"2506.12622","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/dr-sac-distributionally-robust-soft-actor#ran","syntology_url":"https://syntology.ai/paper/2506.12622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12622"}},"official":{"repos":["lemutisme/dr-sac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/treerl-llm-reinforcement-learning-with-on","slug":"treerl-llm-reinforcement-learning-with-on","title":"TreeRL: LLM Reinforcement Learning with On-Policy Tree Search","date":"2025-06-13","arxiv_id":"2506.11902","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/treerl-llm-reinforcement-learning-with-on#ran","syntology_url":"https://syntology.ai/paper/2506.11902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.11902"}},"official":{"repos":["thudm/treerl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/shapley-machine-a-game-theoretic-framework","slug":"shapley-machine-a-game-theoretic-framework","title":"Shapley Machine: A Game-Theoretic Framework for N-Agent Ad Hoc Teamwork","date":"2025-06-12","arxiv_id":"2506.11285","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shapley-machine-a-game-theoretic-framework#ran","syntology_url":"https://syntology.ai/paper/2506.11285","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.11285"}},"official":{"repos":["hsvgbkhgbv/shapley-machine-naht"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-teachers-of-test-time","slug":"reinforcement-learning-teachers-of-test-time","title":"Reinforcement Learning Teachers of Test Time Scaling","date":"2025-06-10","arxiv_id":"2506.08388","repositories_listed":0,"syntology":{"n":25,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":0,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/reinforcement-learning-teachers-of-test-time#ran","syntology_url":"https://syntology.ai/paper/2506.08388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08388"}},"official":null}},{"url":"/paper/offline-rl-with-smooth-ood-generalization-in","slug":"offline-rl-with-smooth-ood-generalization-in","title":"Offline RL with Smooth OOD Generalization in Convex Hull and its Neighborhood","date":"2025-06-10","arxiv_id":"2506.08417","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-rl-with-smooth-ood-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2506.08417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08417"}},"official":{"repos":["yqpqry/sqog"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rulereasoner-reinforced-rule-based-reasoning","slug":"rulereasoner-reinforced-rule-based-reasoning","title":"RuleReasoner: Reinforced Rule-based Reasoning via Domain-aware Dynamic Sampling","date":"2025-06-10","arxiv_id":"2506.08672","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/rulereasoner-reinforced-rule-based-reasoning#ran","syntology_url":"https://syntology.ai/paper/2506.08672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08672"}},"official":{"repos":["bigai-nlco/rulereasoner"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/intention-conditioned-flow-occupancy-models","slug":"intention-conditioned-flow-occupancy-models","title":"Intention-Conditioned Flow Occupancy Models","date":"2025-06-10","arxiv_id":"2506.08902","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":10,"n_ran_checked":10,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 10 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/intention-conditioned-flow-occupancy-models#ran","syntology_url":"https://syntology.ai/paper/2506.08902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08902"}},"official":{"repos":["chongyi-zheng/infom"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":10,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/router-r1-teaching-llms-multi-round-routing","slug":"router-r1-teaching-llms-multi-round-routing","title":"Router-R1: Teaching LLMs Multi-Round Routing and Aggregation via Reinforcement Learning","date":"2025-06-10","arxiv_id":"2506.09033","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/router-r1-teaching-llms-multi-round-routing#ran","syntology_url":"https://syntology.ai/paper/2506.09033","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09033"}},"official":{"repos":["ulab-uiuc/router-r1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/play-to-generalize-learning-to-reason-through","slug":"play-to-generalize-learning-to-reason-through","title":"Play to Generalize: Learning to Reason Through Game Play","date":"2025-06-09","arxiv_id":"2506.08011","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/play-to-generalize-learning-to-reason-through#ran","syntology_url":"https://syntology.ai/paper/2506.08011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08011"}},"official":{"repos":["yunfeixie233/vigal"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/gradual-transition-from-bellman-optimality","slug":"gradual-transition-from-bellman-optimality","title":"Gradual Transition from Bellman Optimality Operator to Bellman Operator in Online Reinforcement Learning","date":"2025-06-06","arxiv_id":"2506.05968","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/gradual-transition-from-bellman-optimality#ran","syntology_url":"https://syntology.ai/paper/2506.05968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.05968"}},"official":{"repos":["motokiomura/annealed-q-learning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-data-efficiency-for-llm","slug":"improving-data-efficiency-for-llm","title":"Improving Data Efficiency for LLM Reinforcement Fine-tuning Through Difficulty-targeted Online Data Selection and Rollout Replay","date":"2025-06-05","arxiv_id":"2506.05316","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-data-efficiency-for-llm#ran","syntology_url":"https://syntology.ai/paper/2506.05316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.05316"}},"official":{"repos":["astral-group/data-efficient-llm-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/latent-guided-sampling-for-combinatorial","slug":"latent-guided-sampling-for-combinatorial","title":"Latent Guided Sampling for Combinatorial Optimization","date":"2025-06-04","arxiv_id":"2506.03672","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-guided-sampling-for-combinatorial#ran","syntology_url":"https://syntology.ai/paper/2506.03672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.03672"}},"official":{"repos":["SobihanSurendran/LGS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/incentivizing-reasoning-for-advanced","slug":"incentivizing-reasoning-for-advanced","title":"Incentivizing Reasoning for Advanced Instruction-Following of Large Language Models","date":"2025-06-02","arxiv_id":"2506.01413","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/incentivizing-reasoning-for-advanced#ran","syntology_url":"https://syntology.ai/paper/2506.01413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01413"}},"official":{"repos":["yuleiqin/raif"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/areal-a-large-scale-asynchronous","slug":"areal-a-large-scale-asynchronous","title":"AReaL: A Large-Scale Asynchronous Reinforcement Learning System for Language Reasoning","date":"2025-05-30","arxiv_id":"2505.24298","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/areal-a-large-scale-asynchronous#ran","syntology_url":"https://syntology.ai/paper/2505.24298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24298"}},"official":{"repos":["inclusionai/areal"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/the-hallucination-dilemma-factuality-aware","slug":"the-hallucination-dilemma-factuality-aware","title":"The Hallucination Dilemma: Factuality-Aware Reinforcement Learning for Large Reasoning Models","date":"2025-05-30","arxiv_id":"2505.24630","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-hallucination-dilemma-factuality-aware#ran","syntology_url":"https://syntology.ai/paper/2505.24630","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24630"}},"official":{"repos":["nusnlp/fspo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/prorl-prolonged-reinforcement-learning","slug":"prorl-prolonged-reinforcement-learning","title":"ProRL: Prolonged Reinforcement Learning Expands Reasoning Boundaries in Large Language Models","date":"2025-05-30","arxiv_id":"2505.24864","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prorl-prolonged-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2505.24864","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24864"}},"official":{"repos":["open-thought/reasoning-gym"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reasongen-r1-cot-for-autoregressive-image","slug":"reasongen-r1-cot-for-autoregressive-image","title":"ReasonGen-R1: CoT for Autoregressive Image generation models through SFT and RL","date":"2025-05-30","arxiv_id":"2505.24875","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/reasongen-r1-cot-for-autoregressive-image#ran","syntology_url":"https://syntology.ai/paper/2505.24875","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24875"}},"official":null}},{"url":"/paper/segment-policy-optimization-effective-segment","slug":"segment-policy-optimization-effective-segment","title":"Segment Policy Optimization: Effective Segment-Level Credit Assignment in RL for Large Language Models","date":"2025-05-29","arxiv_id":"2505.23564","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/segment-policy-optimization-effective-segment#ran","syntology_url":"https://syntology.ai/paper/2505.23564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23564"}},"official":{"repos":["aiframeresearch/spo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/skywork-open-reasoner-1-technical-report","slug":"skywork-open-reasoner-1-technical-report","title":"Skywork Open Reasoner 1 Technical Report","date":"2025-05-28","arxiv_id":"2505.22312","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/skywork-open-reasoner-1-technical-report#ran","syntology_url":"https://syntology.ai/paper/2505.22312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22312"}},"official":{"repos":["skyworkai/skywork-or1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-post-training-for-multi-modal","slug":"unsupervised-post-training-for-multi-modal","title":"Unsupervised Post-Training for Multi-Modal LLM Reasoning via GRPO","date":"2025-05-28","arxiv_id":"2505.22453","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unsupervised-post-training-for-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2505.22453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22453"}},"official":{"repos":["hiyouga/easyr1","waltonfuture/mm-upt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cadrille-multi-modal-cad-reconstruction-with","slug":"cadrille-multi-modal-cad-reconstruction-with","title":"cadrille: Multi-modal CAD Reconstruction with Online Reinforcement Learning","date":"2025-05-28","arxiv_id":"2505.22914","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/cadrille-multi-modal-cad-reconstruction-with#ran","syntology_url":"https://syntology.ai/paper/2505.22914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22914"}},"official":null}},{"url":"/paper/spa-rl-reinforcing-llm-agents-via-stepwise","slug":"spa-rl-reinforcing-llm-agents-via-stepwise","title":"SPA-RL: Reinforcing LLM Agents via Stepwise Progress Attribution","date":"2025-05-27","arxiv_id":"2505.20732","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spa-rl-reinforcing-llm-agents-via-stepwise#ran","syntology_url":"https://syntology.ai/paper/2505.20732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20732"}},"official":{"repos":["wanghanlinhenry/spa-rl-agent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcing-general-reasoning-without","slug":"reinforcing-general-reasoning-without","title":"Reinforcing General Reasoning without Verifiers","date":"2025-05-27","arxiv_id":"2505.21493","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcing-general-reasoning-without#ran","syntology_url":"https://syntology.ai/paper/2505.21493","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21493"}},"official":{"repos":["sail-sg/verifree"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-trust-bellman-updates-selective","slug":"learning-to-trust-bellman-updates-selective","title":"Learning to Trust Bellman Updates: Selective State-Adaptive Regularization for Offline RL","date":"2025-05-26","arxiv_id":"2505.19923","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":2,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":2,"n_pointer_only":7,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 2 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-trust-bellman-updates-selective#ran","syntology_url":"https://syntology.ai/paper/2505.19923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19923"}},"official":{"repos":["qinwenluo/ssar"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/serl-self-play-reinforcement-learning-for","slug":"serl-self-play-reinforcement-learning-for","title":"SeRL: Self-Play Reinforcement Learning for Large Language Models with Limited Data","date":"2025-05-25","arxiv_id":"2505.20347","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/serl-self-play-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2505.20347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20347"}},"official":{"repos":["wantbook-book/serl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-blend-inference-time-multi","slug":"diffusion-blend-inference-time-multi","title":"Diffusion Blend: Inference-Time Multi-Preference Alignment for Diffusion Models","date":"2025-05-24","arxiv_id":"2505.18547","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffusion-blend-inference-time-multi#ran","syntology_url":"https://syntology.ai/paper/2505.18547","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18547"}},"official":{"repos":["bluewoods127/db-2025"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vla-rl-towards-masterful-and-general-robotic","slug":"vla-rl-towards-masterful-and-general-robotic","title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","date":"2025-05-24","arxiv_id":"2505.18719","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vla-rl-towards-masterful-and-general-robotic#ran","syntology_url":"https://syntology.ai/paper/2505.18719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18719"}},"official":{"repos":["guanxinglu/vlarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adactrl-towards-adaptive-and-controllable","slug":"adactrl-towards-adaptive-and-controllable","title":"AdaCtrl: Towards Adaptive and Controllable Reasoning via Difficulty-Aware Budgeting","date":"2025-05-24","arxiv_id":"2505.18822","repositories_listed":1,"syntology":{"n":15,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/adactrl-towards-adaptive-and-controllable#ran","syntology_url":"https://syntology.ai/paper/2505.18822","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18822"}},"official":{"repos":["joeying1019/adactrl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/co-reinforcement-learning-for-unified","slug":"co-reinforcement-learning-for-unified","title":"Co-Reinforcement Learning for Unified Multimodal Understanding and Generation","date":"2025-05-23","arxiv_id":"2505.17534","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-reinforcement-learning-for-unified#ran","syntology_url":"https://syntology.ai/paper/2505.17534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17534"}},"official":{"repos":["mm-vl/ulm-r1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/thinking-fast-and-right-balancing-accuracy","slug":"thinking-fast-and-right-balancing-accuracy","title":"Thinking Fast and Right: Balancing Accuracy and Reasoning Length with Adaptive Rewards","date":"2025-05-23","arxiv_id":"2505.18298","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/thinking-fast-and-right-balancing-accuracy#ran","syntology_url":"https://syntology.ai/paper/2505.18298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18298"}},"official":{"repos":["jinyansu1/a-dlp"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/arpo-end-to-end-policy-optimization-for-gui","slug":"arpo-end-to-end-policy-optimization-for-gui","title":"ARPO:End-to-End Policy Optimization for GUI Agents with Experience Replay","date":"2025-05-22","arxiv_id":"2505.16282","repositories_listed":1,"syntology":{"n":16,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/arpo-end-to-end-policy-optimization-for-gui#ran","syntology_url":"https://syntology.ai/paper/2505.16282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16282"}},"official":{"repos":["dvlab-research/arpo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/saturn-sat-based-reinforcement-learning-to","slug":"saturn-sat-based-reinforcement-learning-to","title":"SATURN: SAT-based Reinforcement Learning to Unleash Language Model Reasoning","date":"2025-05-22","arxiv_id":"2505.16368","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/saturn-sat-based-reinforcement-learning-to#ran","syntology_url":"https://syntology.ai/paper/2505.16368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16368"}},"official":{"repos":["gtxygyzb/saturn-code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"url":"/paper/webagent-r1-training-web-agents-via-end-to","slug":"webagent-r1-training-web-agents-via-end-to","title":"WebAgent-R1: Training Web Agents via End-to-End Multi-Turn Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.16421","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/webagent-r1-training-web-agents-via-end-to#ran","syntology_url":"https://syntology.ai/paper/2505.16421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16421"}},"official":{"repos":["weizhepei/webagent-r1"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/r1-sharevl-incentivizing-reasoning-capability","slug":"r1-sharevl-incentivizing-reasoning-capability","title":"R1-ShareVL: Incentivizing Reasoning Capability of Multimodal Large Language Models via Share-GRPO","date":"2025-05-22","arxiv_id":"2505.16673","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r1-sharevl-incentivizing-reasoning-capability#ran","syntology_url":"https://syntology.ai/paper/2505.16673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16673"}},"official":{"repos":["hjyao00/r1-sharevl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/swe-dev-evaluating-and-training-autonomous","slug":"swe-dev-evaluating-and-training-autonomous","title":"SWE-Dev: Evaluating and Training Autonomous Feature-Driven Software Development","date":"2025-05-22","arxiv_id":"2505.16975","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/swe-dev-evaluating-and-training-autonomous#ran","syntology_url":"https://syntology.ai/paper/2505.16975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16975"}},"official":{"repos":["dorothyduuu/swe-dev"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sophiavl-r1-reinforcing-mllms-reasoning-with","slug":"sophiavl-r1-reinforcing-mllms-reasoning-with","title":"SophiaVL-R1: Reinforcing MLLMs Reasoning with Thinking Reward","date":"2025-05-22","arxiv_id":"2505.17018","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":9,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 9 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sophiavl-r1-reinforcing-mllms-reasoning-with#ran","syntology_url":"https://syntology.ai/paper/2505.17018","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17018"}},"official":{"repos":["hiyouga/easyr1","kxfan2002/sophiavl-r1"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rl-tango-reinforcing-generator-and-verifier","slug":"rl-tango-reinforcing-generator-and-verifier","title":"RL Tango: Reinforcing Generator and Verifier Together for Language Reasoning","date":"2025-05-21","arxiv_id":"2505.15034","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rl-tango-reinforcing-generator-and-verifier#ran","syntology_url":"https://syntology.ai/paper/2505.15034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15034"}},"official":{"repos":["kaiwenzha/rl-tango"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-on-reinforcement-learning","slug":"an-empirical-study-on-reinforcement-learning","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","date":"2025-05-21","arxiv_id":"2505.15117","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-on-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2505.15117","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15117"}},"official":{"repos":["petergriffinjin/search-r1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/trajectory-bellman-residual-minimization-a","slug":"trajectory-bellman-residual-minimization-a","title":"Trajectory Bellman Residual Minimization: A Simple Value-Based Method for LLM Reasoning","date":"2025-05-21","arxiv_id":"2505.15311","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trajectory-bellman-residual-minimization-a#ran","syntology_url":"https://syntology.ai/paper/2505.15311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15311"}},"official":null}},{"url":"/paper/mmada-multimodal-large-diffusion-language","slug":"mmada-multimodal-large-diffusion-language","title":"MMaDA: Multimodal Large Diffusion Language Models","date":"2025-05-21","arxiv_id":"2505.15809","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mmada-multimodal-large-diffusion-language#ran","syntology_url":"https://syntology.ai/paper/2505.15809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15809"}},"official":{"repos":["gen-verse/mmada"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gui-g1-understanding-r1-zero-like-training","slug":"gui-g1-understanding-r1-zero-like-training","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","date":"2025-05-21","arxiv_id":"2505.15810","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/gui-g1-understanding-r1-zero-like-training#ran","syntology_url":"https://syntology.ai/paper/2505.15810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15810"}},"official":{"repos":["yuqi-zhou/gui-g1"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/s3-you-don-t-need-that-much-data-to-train-a","slug":"s3-you-don-t-need-that-much-data-to-train-a","title":"s3: You Don't Need That Much Data to Train a Search Agent via RL","date":"2025-05-20","arxiv_id":"2505.14146","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":3,"n_ran_checked":3,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/s3-you-don-t-need-that-much-data-to-train-a#ran","syntology_url":"https://syntology.ai/paper/2505.14146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14146"}},"official":{"repos":["pat-jj/s3"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tinyv-reducing-false-negatives-in","slug":"tinyv-reducing-false-negatives-in","title":"TinyV: Reducing False Negatives in Verification Improves RL for LLM Reasoning","date":"2025-05-20","arxiv_id":"2505.14625","repositories_listed":1,"syntology":{"n":19,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/tinyv-reducing-false-negatives-in#ran","syntology_url":"https://syntology.ai/paper/2505.14625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14625"}},"official":{"repos":["uw-nsl/tinyv"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/do-not-let-low-probability-tokens-over","slug":"do-not-let-low-probability-tokens-over","title":"Do Not Let Low-Probability Tokens Over-Dominate in RL for LLMs","date":"2025-05-19","arxiv_id":"2505.12929","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/do-not-let-low-probability-tokens-over#ran","syntology_url":"https://syntology.ai/paper/2505.12929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12929"}},"official":{"repos":["zhyang2226/ar-lopti"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/videorft-incentivizing-video-reasoning","slug":"videorft-incentivizing-video-reasoning","title":"VideoRFT: Incentivizing Video Reasoning Capability in MLLMs via Reinforced Fine-Tuning","date":"2025-05-18","arxiv_id":"2505.12434","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 3 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videorft-incentivizing-video-reasoning#ran","syntology_url":"https://syntology.ai/paper/2505.12434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12434"}},"official":{"repos":["qiwang98/videorft"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/synthetic-data-rl-task-definition-is-all-you","slug":"synthetic-data-rl-task-definition-is-all-you","title":"Synthetic Data RL: Task Definition Is All You Need","date":"2025-05-18","arxiv_id":"2505.17063","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/synthetic-data-rl-task-definition-is-all-you#ran","syntology_url":"https://syntology.ai/paper/2505.17063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17063"}},"official":{"repos":["gydpku/data_synthesis_rl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-10861","slug":"2505-10861","title":"Improving the Data-efficiency of Reinforcement Learning by Warm-starting with LLM","date":"2025-05-16","arxiv_id":"2505.10861","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2505-10861#ran","syntology_url":"https://syntology.ai/paper/2505.10861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10861"}},"official":{"repos":["duongnhatthang/llamagym"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-finetunes-small","slug":"reinforcement-learning-finetunes-small","title":"Reinforcement Learning Finetunes Small Subnetworks in Large Language Models","date":"2025-05-16","arxiv_id":"2505.11711","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-finetunes-small#ran","syntology_url":"https://syntology.ai/paper/2505.11711","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11711"}},"official":null}},{"url":"/paper/tensorrl-qas-reinforcement-learning-with","slug":"tensorrl-qas-reinforcement-learning-with","title":"TensorRL-QAS: Reinforcement learning with tensor networks for scalable quantum architecture search","date":"2025-05-14","arxiv_id":"2505.09371","repositories_listed":0,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tensorrl-qas-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2505.09371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.09371"}},"official":null}},{"url":"/paper/measuring-general-intelligence-with-generated","slug":"measuring-general-intelligence-with-generated","title":"Measuring General Intelligence with Generated Games","date":"2025-05-12","arxiv_id":"2505.07215","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/measuring-general-intelligence-with-generated#ran","syntology_url":"https://syntology.ai/paper/2505.07215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07215"}},"official":{"repos":["vivek3141/gg-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforced-internal-external-knowledge","slug":"reinforced-internal-external-knowledge","title":"Reinforced Internal-External Knowledge Synergistic Reasoning for Efficient Adaptive Search Agent","date":"2025-05-12","arxiv_id":"2505.07596","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforced-internal-external-knowledge#ran","syntology_url":"https://syntology.ai/paper/2505.07596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07596"}},"official":null}},{"url":"/paper/agent-rl-scaling-law-agent-rl-with","slug":"agent-rl-scaling-law-agent-rl-with","title":"Agent RL Scaling Law: Agent RL with Spontaneous Code Execution for Mathematical Problem Solving","date":"2025-05-12","arxiv_id":"2505.07773","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agent-rl-scaling-law-agent-rl-with#ran","syntology_url":"https://syntology.ai/paper/2505.07773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07773"}},"official":{"repos":["anonymize-author/agentrl","yyht/openrlhf_async_pipline"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dancegrpo-unleashing-grpo-on-visual","slug":"dancegrpo-unleashing-grpo-on-visual","title":"DanceGRPO: Unleashing GRPO on Visual Generation","date":"2025-05-12","arxiv_id":"2505.07818","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dancegrpo-unleashing-grpo-on-visual#ran","syntology_url":"https://syntology.ai/paper/2505.07818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07818"}},"official":null}},{"url":"/paper/lineflow-a-framework-to-learn-active-control","slug":"lineflow-a-framework-to-learn-active-control","title":"LineFlow: A Framework to Learn Active Control of Production Lines","date":"2025-05-10","arxiv_id":"2505.06744","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lineflow-a-framework-to-learn-active-control#ran","syntology_url":"https://syntology.ai/paper/2505.06744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.06744"}},"official":{"repos":["hs-kempten/lineflow"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flow-grpo-training-flow-matching-models-via","slug":"flow-grpo-training-flow-matching-models-via","title":"Flow-GRPO: Training Flow Matching Models via Online RL","date":"2025-05-08","arxiv_id":"2505.05470","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flow-grpo-training-flow-matching-models-via#ran","syntology_url":"https://syntology.ai/paper/2505.05470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05470"}},"official":{"repos":["yifan123/flow_grpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/zerosearch-incentivize-the-search-capability","slug":"zerosearch-incentivize-the-search-capability","title":"ZeroSearch: Incentivize the Search Capability of LLMs without Searching","date":"2025-05-07","arxiv_id":"2505.04588","repositories_listed":1,"syntology":{"n":15,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/zerosearch-incentivize-the-search-capability#ran","syntology_url":"https://syntology.ai/paper/2505.04588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.04588"}},"official":{"repos":["alibaba-nlp/zerosearch"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/rm-r1-reward-modeling-as-reasoning","slug":"rm-r1-reward-modeling-as-reasoning","title":"RM-R1: Reward Modeling as Reasoning","date":"2025-05-05","arxiv_id":"2505.02387","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/rm-r1-reward-modeling-as-reasoning#ran","syntology_url":"https://syntology.ai/paper/2505.02387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02387"}},"official":{"repos":["rm-r1-uiuc/rm-r1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/r1-reward-training-multimodal-reward-model","slug":"r1-reward-training-multimodal-reward-model","title":"R1-Reward: Training Multimodal Reward Model Through Stable Reinforcement Learning","date":"2025-05-05","arxiv_id":"2505.02835","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/r1-reward-training-multimodal-reward-model#ran","syntology_url":"https://syntology.ai/paper/2505.02835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02835"}},"official":{"repos":["yfzhang114/r1_reward"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/t2i-r1-reinforcing-image-generation-with","slug":"t2i-r1-reinforcing-image-generation-with","title":"T2I-R1: Reinforcing Image Generation with Collaborative Semantic-level and Token-level CoT","date":"2025-05-01","arxiv_id":"2505.00703","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/t2i-r1-reinforcing-image-generation-with#ran","syntology_url":"https://syntology.ai/paper/2505.00703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.00703"}},"official":{"repos":["caraj7/t2i-r1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/carl-learning-scalable-planning-policies-with","slug":"carl-learning-scalable-planning-policies-with","title":"CaRL: Learning Scalable Planning Policies with Simple Rewards","date":"2025-04-24","arxiv_id":"2504.17838","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/carl-learning-scalable-planning-policies-with#ran","syntology_url":"https://syntology.ai/paper/2504.17838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.17838"}},"official":{"repos":["autonomousvision/CaRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ragen-understanding-self-evolution-in-llm","slug":"ragen-understanding-self-evolution-in-llm","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","date":"2025-04-24","arxiv_id":"2504.20073","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ragen-understanding-self-evolution-in-llm#ran","syntology_url":"https://syntology.ai/paper/2504.20073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20073"}},"official":{"repos":["ragen-ai/ragen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tina-tiny-reasoning-models-via-lora","slug":"tina-tiny-reasoning-models-via-lora","title":"Tina: Tiny Reasoning Models via LoRA","date":"2025-04-22","arxiv_id":"2504.15777","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tina-tiny-reasoning-models-via-lora#ran","syntology_url":"https://syntology.ai/paper/2504.15777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15777"}},"official":{"repos":["shangshang-wang/tina"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ttrl-test-time-reinforcement-learning","slug":"ttrl-test-time-reinforcement-learning","title":"TTRL: Test-Time Reinforcement Learning","date":"2025-04-22","arxiv_id":"2504.16084","repositories_listed":3,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":6,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ttrl-test-time-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2504.16084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16084"}},"official":{"repos":["prime-rl/ttrl","tsinghuac3i/awesome-rl-reasoning-recipes"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/stop-summation-min-form-credit-assignment-is","slug":"stop-summation-min-form-credit-assignment-is","title":"Stop Summation: Min-Form Credit Assignment Is All Process Reward Model Needs for Reasoning","date":"2025-04-21","arxiv_id":"2504.15275","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stop-summation-min-form-credit-assignment-is#ran","syntology_url":"https://syntology.ai/paper/2504.15275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15275"}},"official":{"repos":["cjreinforce/pure"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-auto-bidding-with-value-guided","slug":"generative-auto-bidding-with-value-guided","title":"Generative Auto-Bidding with Value-Guided Explorations","date":"2025-04-20","arxiv_id":"2504.14587","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generative-auto-bidding-with-value-guided#ran","syntology_url":"https://syntology.ai/paper/2504.14587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.14587"}},"official":{"repos":["applied-machine-learning-lab/gave"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/compile-scene-graphs-with-reinforcement","slug":"compile-scene-graphs-with-reinforcement","title":"Compile Scene Graphs with Reinforcement Learning","date":"2025-04-18","arxiv_id":"2504.13617","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/compile-scene-graphs-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2504.13617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13617"}},"official":{"repos":["gpt4vision/r1-sgg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/noisyrollout-reinforcing-visual-reasoning","slug":"noisyrollout-reinforcing-visual-reasoning","title":"NoisyRollout: Reinforcing Visual Reasoning with Data Augmentation","date":"2025-04-17","arxiv_id":"2504.13055","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/noisyrollout-reinforcing-visual-reasoning#ran","syntology_url":"https://syntology.ai/paper/2504.13055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13055"}},"official":null}},{"url":"/paper/vipo-value-function-inconsistency-penalized","slug":"vipo-value-function-inconsistency-penalized","title":"VIPO: Value Function Inconsistency Penalized Offline Reinforcement Learning","date":"2025-04-16","arxiv_id":"2504.11944","repositories_listed":0,"syntology":{"n":14,"n_ran":10,"n_constructed":9,"n_ran_checked":9,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":14,"phrase":"10 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vipo-value-function-inconsistency-penalized#ran","syntology_url":"https://syntology.ai/paper/2504.11944","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11944"}},"official":null}},{"url":"/paper/toolrl-reward-is-all-tool-learning-needs","slug":"toolrl-reward-is-all-tool-learning-needs","title":"ToolRL: Reward is All Tool Learning Needs","date":"2025-04-16","arxiv_id":"2504.13958","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/toolrl-reward-is-all-tool-learning-needs#ran","syntology_url":"https://syntology.ai/paper/2504.13958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13958"}},"official":{"repos":["qiancheng0/toolrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-whole-body-humanoid-control-via","slug":"zero-shot-whole-body-humanoid-control-via","title":"Zero-Shot Whole-Body Humanoid Control via Behavioral Foundation Models","date":"2025-04-15","arxiv_id":"2504.11054","repositories_listed":2,"syntology":{"n":31,"n_ran":16,"n_constructed":9,"n_ran_checked":10,"n_instrument":6,"n_unverified":15,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":31,"phrase":"16 ran (of which 9 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 15 unverified","sample_list":"/paper/zero-shot-whole-body-humanoid-control-via#ran","syntology_url":"https://syntology.ai/paper/2504.11054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11054"}},"official":{"repos":["facebookresearch/humenv","facebookresearch/metamotivo"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":9,"n_ran_no_instrument_failure":10,"n_unverified":15,"ran_from_kinds":["official"]}}},{"url":"/paper/a-minimalist-approach-to-llm-reasoning-from","slug":"a-minimalist-approach-to-llm-reasoning-from","title":"A Minimalist Approach to LLM Reasoning: from Rejection Sampling to Reinforce","date":"2025-04-15","arxiv_id":"2504.11343","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-minimalist-approach-to-llm-reasoning-from#ran","syntology_url":"https://syntology.ai/paper/2504.11343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11343"}},"official":{"repos":["rlhflow/minimal-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepmath-103k-a-large-scale-challenging","slug":"deepmath-103k-a-large-scale-challenging","title":"DeepMath-103K: A Large-Scale, Challenging, Decontaminated, and Verifiable Mathematical Dataset for Advancing Reasoning","date":"2025-04-15","arxiv_id":"2504.11456","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepmath-103k-a-large-scale-challenging#ran","syntology_url":"https://syntology.ai/paper/2504.11456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11456"}},"official":{"repos":["zwhe99/deepmath"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mt-r1-zero-advancing-llm-based-machine","slug":"mt-r1-zero-advancing-llm-based-machine","title":"MT-R1-Zero: Advancing LLM-based Machine Translation via R1-Zero-like Reinforcement Learning","date":"2025-04-14","arxiv_id":"2504.10160","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mt-r1-zero-advancing-llm-based-machine#ran","syntology_url":"https://syntology.ai/paper/2504.10160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10160"}},"official":{"repos":["fzp0424/mt-r1-zero"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/echo-chamber-rl-post-training-amplifies","slug":"echo-chamber-rl-post-training-amplifies","title":"Echo Chamber: RL Post-training Amplifies Behaviors Learned in Pretraining","date":"2025-04-10","arxiv_id":"2504.07912","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/echo-chamber-rl-post-training-amplifies#ran","syntology_url":"https://syntology.ai/paper/2504.07912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07912"}},"official":{"repos":["rosieyzh/openrlhf-pretrain"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/perception-r1-pioneering-perception-policy","slug":"perception-r1-pioneering-perception-policy","title":"Perception-R1: Pioneering Perception Policy with Reinforcement Learning","date":"2025-04-10","arxiv_id":"2504.07954","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/perception-r1-pioneering-perception-policy#ran","syntology_url":"https://syntology.ai/paper/2504.07954","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07954"}},"official":{"repos":["linkangheng/pr1"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/right-question-is-already-half-the-answer","slug":"right-question-is-already-half-the-answer","title":"Right Question is Already Half the Answer: Fully Unsupervised LLM Reasoning Incentivization","date":"2025-04-08","arxiv_id":"2504.05812","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/right-question-is-already-half-the-answer#ran","syntology_url":"https://syntology.ai/paper/2504.05812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05812"}},"official":{"repos":["qingyangzhang/empo"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/concise-reasoning-via-reinforcement-learning","slug":"concise-reasoning-via-reinforcement-learning","title":"Concise Reasoning via Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.05185","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/concise-reasoning-via-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2504.05185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05185"}},"official":{"repos":["ai-wand/concise-reasoning"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/deepresearcher-scaling-deep-research-via","slug":"deepresearcher-scaling-deep-research-via","title":"DeepResearcher: Scaling Deep Research via Reinforcement Learning in Real-world Environments","date":"2025-04-04","arxiv_id":"2504.03160","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepresearcher-scaling-deep-research-via#ran","syntology_url":"https://syntology.ai/paper/2504.03160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.03160"}},"official":{"repos":["gair-nlp/deepresearcher"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gpg-a-simple-and-strong-reinforcement","slug":"gpg-a-simple-and-strong-reinforcement","title":"GPG: A Simple and Strong Reinforcement Learning Baseline for Model Reasoning","date":"2025-04-03","arxiv_id":"2504.02546","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpg-a-simple-and-strong-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2504.02546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02546"}},"official":{"repos":["amap-ml/gpg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-swe-bench-a-multilingual-benchmark-for","slug":"multi-swe-bench-a-multilingual-benchmark-for","title":"Multi-SWE-bench: A Multilingual Benchmark for Issue Resolving","date":"2025-04-03","arxiv_id":"2504.02605","repositories_listed":3,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-swe-bench-a-multilingual-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2504.02605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02605"}},"official":{"repos":["multi-swe-bench/experiments","multi-swe-bench/mopenhands","multi-swe-bench/multi-swe-bench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/thinkprune-pruning-long-chain-of-thought-of","slug":"thinkprune-pruning-long-chain-of-thought-of","title":"ThinkPrune: Pruning Long Chain-of-Thought of LLMs via Reinforcement Learning","date":"2025-04-02","arxiv_id":"2504.01296","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/thinkprune-pruning-long-chain-of-thought-of#ran","syntology_url":"https://syntology.ai/paper/2504.01296","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.01296"}},"official":{"repos":["UCSB-NLP-Chang/ThinkPrune"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/tom-rl-reinforcement-learning-unlocks-theory","slug":"tom-rl-reinforcement-learning-unlocks-theory","title":"Do Theory of Mind Benchmarks Need Explicit Human-like Reasoning in Language Models?","date":"2025-04-02","arxiv_id":"2504.01698","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/tom-rl-reinforcement-learning-unlocks-theory#ran","syntology_url":"https://syntology.ai/paper/2504.01698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.01698"}},"official":{"repos":["bigai-ai/ToM-RL"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-the-effect-of-reinforcement","slug":"exploring-the-effect-of-reinforcement","title":"Exploring the Effect of Reinforcement Learning on Video Understanding: Insights from SEED-Bench-R1","date":"2025-03-31","arxiv_id":"2503.24376","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/exploring-the-effect-of-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2503.24376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.24376"}},"official":{"repos":["tencentarc/seed-bench-r1"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"45690c9bb86688131ea53bd4923289e0b584e69f7f54fc49cb512b0ae0813c45","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}