{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/11","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":152,"rows_per_page":100,"rows":[1001,1100],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/10","next":"/task/reinforcement-learning-1/papers/12","papers":[{"url":"/paper/shapley-machine-a-game-theoretic-framework","slug":"shapley-machine-a-game-theoretic-framework","title":"Shapley Machine: A Game-Theoretic Framework for N-Agent Ad Hoc Teamwork","date":"2025-06-12","arxiv_id":"2506.11285","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shapley-machine-a-game-theoretic-framework#ran","syntology_url":"https://syntology.ai/paper/2506.11285","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.11285"}},"official":{"repos":["hsvgbkhgbv/shapley-machine-naht"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/viability-of-future-actions-robust-safety-in","slug":"viability-of-future-actions-robust-safety-in","title":"Viability of Future Actions: Robust Safety in Reinforcement Learning via Entropy Regularization","date":"2025-06-12","arxiv_id":"2506.10871","repositories_listed":1,"syntology":null},{"url":"/paper/repo-replay-enhanced-policy-optimization","slug":"repo-replay-enhanced-policy-optimization","title":"RePO: Replay-Enhanced Policy Optimization","date":"2025-06-11","arxiv_id":"2506.09340","repositories_listed":1,"syntology":null},{"url":"/paper/vicrit-a-verifiable-reinforcement-learning","slug":"vicrit-a-verifiable-reinforcement-learning","title":"ViCrit: A Verifiable Reinforcement Learning Proxy Task for Visual Perception in VLMs","date":"2025-06-11","arxiv_id":"2506.10128","repositories_listed":1,"syntology":null},{"url":"/paper/consistent-paths-lead-to-truth-self-rewarding","slug":"consistent-paths-lead-to-truth-self-rewarding","title":"Consistent Paths Lead to Truth: Self-Rewarding Reinforcement Learning for LLM Reasoning","date":"2025-06-10","arxiv_id":"2506.08745","repositories_listed":1,"syntology":null},{"url":"/paper/intention-conditioned-flow-occupancy-models","slug":"intention-conditioned-flow-occupancy-models","title":"Intention-Conditioned Flow Occupancy Models","date":"2025-06-10","arxiv_id":"2506.08902","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":10,"n_ran_checked":10,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 10 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/intention-conditioned-flow-occupancy-models#ran","syntology_url":"https://syntology.ai/paper/2506.08902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08902"}},"official":{"repos":["chongyi-zheng/infom"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":10,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-rl-with-smooth-ood-generalization-in","slug":"offline-rl-with-smooth-ood-generalization-in","title":"Offline RL with Smooth OOD Generalization in Convex Hull and its Neighborhood","date":"2025-06-10","arxiv_id":"2506.08417","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-rl-with-smooth-ood-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2506.08417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08417"}},"official":{"repos":["yqpqry/sqog"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/router-r1-teaching-llms-multi-round-routing","slug":"router-r1-teaching-llms-multi-round-routing","title":"Router-R1: Teaching LLMs Multi-Round Routing and Aggregation via Reinforcement Learning","date":"2025-06-10","arxiv_id":"2506.09033","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/router-r1-teaching-llms-multi-round-routing#ran","syntology_url":"https://syntology.ai/paper/2506.09033","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09033"}},"official":{"repos":["ulab-uiuc/router-r1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rulereasoner-reinforced-rule-based-reasoning","slug":"rulereasoner-reinforced-rule-based-reasoning","title":"RuleReasoner: Reinforced Rule-based Reasoning via Domain-aware Dynamic Sampling","date":"2025-06-10","arxiv_id":"2506.08672","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/rulereasoner-reinforced-rule-based-reasoning#ran","syntology_url":"https://syntology.ai/paper/2506.08672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08672"}},"official":{"repos":["bigai-nlco/rulereasoner"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/speed-rl-faster-training-of-reasoning-models","slug":"speed-rl-faster-training-of-reasoning-models","title":"SPEED-RL: Faster Training of Reasoning Models via Online Curriculum Learning","date":"2025-06-10","arxiv_id":"2506.09016","repositories_listed":1,"syntology":null},{"url":"/paper/compound-ai-systems-optimization-a-survey-of","slug":"compound-ai-systems-optimization-a-survey-of","title":"Compound AI Systems Optimization: A Survey of Methods, Challenges, and Future Directions","date":"2025-06-09","arxiv_id":"2506.08234","repositories_listed":1,"syntology":null},{"url":"/paper/learning-what-reinforcement-learning-can-t","slug":"learning-what-reinforcement-learning-can-t","title":"Learning What Reinforcement Learning Can't: Interleaved Online Fine-Tuning for Hardest Questions","date":"2025-06-09","arxiv_id":"2506.07527","repositories_listed":1,"syntology":null},{"url":"/paper/play-to-generalize-learning-to-reason-through","slug":"play-to-generalize-learning-to-reason-through","title":"Play to Generalize: Learning to Reason Through Game Play","date":"2025-06-09","arxiv_id":"2506.08011","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/play-to-generalize-learning-to-reason-through#ran","syntology_url":"https://syntology.ai/paper/2506.08011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08011"}},"official":{"repos":["yunfeixie233/vigal"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/thinking-vs-doing-agents-that-reason-by","slug":"thinking-vs-doing-agents-that-reason-by","title":"Thinking vs. Doing: Agents that Reason by Scaling Test-Time Interaction","date":"2025-06-09","arxiv_id":"2506.07976","repositories_listed":1,"syntology":null},{"url":"/paper/wethink-toward-general-purpose-vision","slug":"wethink-toward-general-purpose-vision","title":"WeThink: Toward General-purpose Vision-Language Reasoning via Reinforcement Learning","date":"2025-06-09","arxiv_id":"2506.07905","repositories_listed":1,"syntology":null},{"url":"/paper/gradual-transition-from-bellman-optimality","slug":"gradual-transition-from-bellman-optimality","title":"Gradual Transition from Bellman Optimality Operator to Bellman Operator in Online Reinforcement Learning","date":"2025-06-06","arxiv_id":"2506.05968","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/gradual-transition-from-bellman-optimality#ran","syntology_url":"https://syntology.ai/paper/2506.05968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.05968"}},"official":{"repos":["motokiomura/annealed-q-learning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/dissecting-long-reasoning-models-an-empirical","slug":"dissecting-long-reasoning-models-an-empirical","title":"Dissecting Long Reasoning Models: An Empirical Study","date":"2025-06-05","arxiv_id":"2506.04913","repositories_listed":1,"syntology":null},{"url":"/paper/improving-data-efficiency-for-llm","slug":"improving-data-efficiency-for-llm","title":"Improving Data Efficiency for LLM Reinforcement Fine-tuning Through Difficulty-targeted Online Data Selection and Rollout Replay","date":"2025-06-05","arxiv_id":"2506.05316","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-data-efficiency-for-llm#ran","syntology_url":"https://syntology.ai/paper/2506.05316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.05316"}},"official":{"repos":["astral-group/data-efficient-llm-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/latent-guided-sampling-for-combinatorial","slug":"latent-guided-sampling-for-combinatorial","title":"Latent Guided Sampling for Combinatorial Optimization","date":"2025-06-04","arxiv_id":"2506.03672","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-guided-sampling-for-combinatorial#ran","syntology_url":"https://syntology.ai/paper/2506.03672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.03672"}},"official":{"repos":["SobihanSurendran/LGS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/incentivizing-reasoning-for-advanced","slug":"incentivizing-reasoning-for-advanced","title":"Incentivizing Reasoning for Advanced Instruction-Following of Large Language Models","date":"2025-06-02","arxiv_id":"2506.01413","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/incentivizing-reasoning-for-advanced#ran","syntology_url":"https://syntology.ai/paper/2506.01413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01413"}},"official":{"repos":["yuleiqin/raif"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reasoning-table-exploring-reinforcement","slug":"reasoning-table-exploring-reinforcement","title":"Reasoning-Table: Exploring Reinforcement Learning for Table Reasoning","date":"2025-06-02","arxiv_id":"2506.01710","repositories_listed":1,"syntology":null},{"url":"/paper/areal-a-large-scale-asynchronous","slug":"areal-a-large-scale-asynchronous","title":"AReaL: A Large-Scale Asynchronous Reinforcement Learning System for Language Reasoning","date":"2025-05-30","arxiv_id":"2505.24298","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/areal-a-large-scale-asynchronous#ran","syntology_url":"https://syntology.ai/paper/2505.24298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24298"}},"official":{"repos":["inclusionai/areal"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mixed-r1-unified-reward-perspective-for","slug":"mixed-r1-unified-reward-perspective-for","title":"Mixed-R1: Unified Reward Perspective For Reasoning Capability in Multimodal Large Language Models","date":"2025-05-30","arxiv_id":"2505.24164","repositories_listed":1,"syntology":null},{"url":"/paper/mofgpt-generative-design-of-metal-organic","slug":"mofgpt-generative-design-of-metal-organic","title":"MOFGPT: Generative Design of Metal-Organic Frameworks using Language Models","date":"2025-05-30","arxiv_id":"2506.00198","repositories_listed":1,"syntology":null},{"url":"/paper/prorl-prolonged-reinforcement-learning","slug":"prorl-prolonged-reinforcement-learning","title":"ProRL: Prolonged Reinforcement Learning Expands Reasoning Boundaries in Large Language Models","date":"2025-05-30","arxiv_id":"2505.24864","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prorl-prolonged-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2505.24864","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24864"}},"official":{"repos":["open-thought/reasoning-gym"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reasongen-r1-cot-for-autoregressive-image","slug":"reasongen-r1-cot-for-autoregressive-image","title":"ReasonGen-R1: CoT for Autoregressive Image generation models through SFT and RL","date":"2025-05-30","arxiv_id":"2505.24875","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/reasongen-r1-cot-for-autoregressive-image#ran","syntology_url":"https://syntology.ai/paper/2505.24875","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24875"}},"official":null}},{"url":"/paper/the-hallucination-dilemma-factuality-aware","slug":"the-hallucination-dilemma-factuality-aware","title":"The Hallucination Dilemma: Factuality-Aware Reinforcement Learning for Large Reasoning Models","date":"2025-05-30","arxiv_id":"2505.24630","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-hallucination-dilemma-factuality-aware#ran","syntology_url":"https://syntology.ai/paper/2505.24630","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24630"}},"official":{"repos":["nusnlp/fspo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/towards-effective-code-integrated-reasoning","slug":"towards-effective-code-integrated-reasoning","title":"Towards Effective Code-Integrated Reasoning","date":"2025-05-30","arxiv_id":"2505.24480","repositories_listed":1,"syntology":null},{"url":"/paper/grounded-reinforcement-learning-for-visual","slug":"grounded-reinforcement-learning-for-visual","title":"Grounded Reinforcement Learning for Visual Reasoning","date":"2025-05-29","arxiv_id":"2505.23678","repositories_listed":1,"syntology":null},{"url":"/paper/jigsaw-r1-a-study-of-rule-based-visual","slug":"jigsaw-r1-a-study-of-rule-based-visual","title":"Jigsaw-R1: A Study of Rule-based Visual Reinforcement Learning with Jigsaw Puzzles","date":"2025-05-29","arxiv_id":"2505.23590","repositories_listed":1,"syntology":null},{"url":"/paper/ml-agent-reinforcing-llm-agents-for","slug":"ml-agent-reinforcing-llm-agents-for","title":"ML-Agent: Reinforcing LLM Agents for Autonomous Machine Learning Engineering","date":"2025-05-29","arxiv_id":"2505.23723","repositories_listed":1,"syntology":null},{"url":"/paper/normalizing-flows-are-capable-models-for-rl","slug":"normalizing-flows-are-capable-models-for-rl","title":"Normalizing Flows are Capable Models for RL","date":"2025-05-29","arxiv_id":"2505.23527","repositories_listed":1,"syntology":null},{"url":"/paper/satori-swe-evolutionary-test-time-scaling-for","slug":"satori-swe-evolutionary-test-time-scaling-for","title":"Satori-SWE: Evolutionary Test-Time Scaling for Sample-Efficient Software Engineering","date":"2025-05-29","arxiv_id":"2505.23604","repositories_listed":1,"syntology":null},{"url":"/paper/segment-policy-optimization-effective-segment","slug":"segment-policy-optimization-effective-segment","title":"Segment Policy Optimization: Effective Segment-Level Credit Assignment in RL for Large Language Models","date":"2025-05-29","arxiv_id":"2505.23564","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/segment-policy-optimization-effective-segment#ran","syntology_url":"https://syntology.ai/paper/2505.23564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23564"}},"official":{"repos":["aiframeresearch/spo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/cadrille-multi-modal-cad-reconstruction-with","slug":"cadrille-multi-modal-cad-reconstruction-with","title":"cadrille: Multi-modal CAD Reconstruction with Online Reinforcement Learning","date":"2025-05-28","arxiv_id":"2505.22914","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/cadrille-multi-modal-cad-reconstruction-with#ran","syntology_url":"https://syntology.ai/paper/2505.22914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22914"}},"official":null}},{"url":"/paper/decomposing-elements-of-problem-solving-what","slug":"decomposing-elements-of-problem-solving-what","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","date":"2025-05-28","arxiv_id":"2505.22756","repositories_listed":1,"syntology":null},{"url":"/paper/hddlgym-a-tool-for-studying-multi-agent","slug":"hddlgym-a-tool-for-studying-multi-agent","title":"HDDLGym: A Tool for Studying Multi-Agent Hierarchical Problems Defined in HDDL with OpenAI Gym","date":"2025-05-28","arxiv_id":"2505.22597","repositories_listed":1,"syntology":null},{"url":"/paper/skywork-open-reasoner-1-technical-report","slug":"skywork-open-reasoner-1-technical-report","title":"Skywork Open Reasoner 1 Technical Report","date":"2025-05-28","arxiv_id":"2505.22312","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/skywork-open-reasoner-1-technical-report#ran","syntology_url":"https://syntology.ai/paper/2505.22312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22312"}},"official":{"repos":["skyworkai/skywork-or1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sorel-and-torel-two-methods-for-fully-offline","slug":"sorel-and-torel-two-methods-for-fully-offline","title":"SOReL and TOReL: Two Methods for Fully Offline Reinforcement Learning","date":"2025-05-28","arxiv_id":"2505.22442","repositories_listed":1,"syntology":null},{"url":"/paper/when-does-neuroevolution-outcompete","slug":"when-does-neuroevolution-outcompete","title":"When Does Neuroevolution Outcompete Reinforcement Learning in Transfer Learning Tasks?","date":"2025-05-28","arxiv_id":"2505.22696","repositories_listed":1,"syntology":null},{"url":"/paper/museg-reinforcing-video-temporal","slug":"museg-reinforcing-video-temporal","title":"MUSEG: Reinforcing Video Temporal Understanding via Timestamp-Aware Multi-Segment Grounding","date":"2025-05-27","arxiv_id":"2505.20715","repositories_listed":1,"syntology":null},{"url":"/paper/r1-code-interpreter-training-llms-to-reason","slug":"r1-code-interpreter-training-llms-to-reason","title":"R1-Code-Interpreter: Training LLMs to Reason with Code via Supervised and Reinforcement Learning","date":"2025-05-27","arxiv_id":"2505.21668","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcing-general-reasoning-without","slug":"reinforcing-general-reasoning-without","title":"Reinforcing General Reasoning without Verifiers","date":"2025-05-27","arxiv_id":"2505.21493","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcing-general-reasoning-without#ran","syntology_url":"https://syntology.ai/paper/2505.21493","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21493"}},"official":{"repos":["sail-sg/verifree"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/spa-rl-reinforcing-llm-agents-via-stepwise","slug":"spa-rl-reinforcing-llm-agents-via-stepwise","title":"SPA-RL: Reinforcing LLM Agents via Stepwise Progress Attribution","date":"2025-05-27","arxiv_id":"2505.20732","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spa-rl-reinforcing-llm-agents-via-stepwise#ran","syntology_url":"https://syntology.ai/paper/2505.20732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20732"}},"official":{"repos":["wanghanlinhenry/spa-rl-agent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ctrl-dna-controllable-cell-type-specific","slug":"ctrl-dna-controllable-cell-type-specific","title":"Ctrl-DNA: Controllable Cell-Type-Specific Regulatory DNA Design via Constrained RL","date":"2025-05-26","arxiv_id":"2505.20578","repositories_listed":1,"syntology":null},{"url":"/paper/discover-automated-curricula-for-sparse","slug":"discover-automated-curricula-for-sparse","title":"DISCOVER: Automated Curricula for Sparse-Reward Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.19850","repositories_listed":1,"syntology":null},{"url":"/paper/doctoragent-rl-a-multi-agent-collaborative","slug":"doctoragent-rl-a-multi-agent-collaborative","title":"DoctorAgent-RL: A Multi-Agent Collaborative Reinforcement Learning System for Multi-Turn Clinical Dialogue","date":"2025-05-26","arxiv_id":"2505.19630","repositories_listed":1,"syntology":null},{"url":"/paper/incentivizing-reasoning-from-weak-supervision","slug":"incentivizing-reasoning-from-weak-supervision","title":"Incentivizing Reasoning from Weak Supervision","date":"2025-05-26","arxiv_id":"2505.20072","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-trust-bellman-updates-selective","slug":"learning-to-trust-bellman-updates-selective","title":"Learning to Trust Bellman Updates: Selective State-Adaptive Regularization for Offline RL","date":"2025-05-26","arxiv_id":"2505.19923","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":2,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":2,"n_pointer_only":7,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 2 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-trust-bellman-updates-selective#ran","syntology_url":"https://syntology.ai/paper/2505.19923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19923"}},"official":{"repos":["qinwenluo/ssar"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/masksearch-a-universal-pre-training-framework","slug":"masksearch-a-universal-pre-training-framework","title":"MASKSEARCH: A Universal Pre-Training Framework to Enhance Agentic Search Capability","date":"2025-05-26","arxiv_id":"2505.20285","repositories_listed":1,"syntology":null},{"url":"/paper/omni-r1-reinforcement-learning-for-omnimodal","slug":"omni-r1-reinforcement-learning-for-omnimodal","title":"Omni-R1: Reinforcement Learning for Omnimodal Reasoning via Two-System Collaboration","date":"2025-05-26","arxiv_id":"2505.20256","repositories_listed":1,"syntology":null},{"url":"/paper/refining-few-step-text-to-multiview-diffusion","slug":"refining-few-step-text-to-multiview-diffusion","title":"Refining Few-Step Text-to-Multiview Diffusion via Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.20107","repositories_listed":1,"syntology":null},{"url":"/paper/synlogic-synthesizing-verifiable-reasoning","slug":"synlogic-synthesizing-verifiable-reasoning","title":"SynLogic: Synthesizing Verifiable Reasoning Data at Scale for Learning Logical Reasoning and Beyond","date":"2025-05-26","arxiv_id":"2505.19641","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-the-compositional-ability-gap-in","slug":"unveiling-the-compositional-ability-gap-in","title":"Unveiling the Compositional Ability Gap in Vision-Language Reasoning Model","date":"2025-05-26","arxiv_id":"2505.19406","repositories_listed":1,"syntology":null},{"url":"/paper/a-snapshot-of-influence-a-local-data","slug":"a-snapshot-of-influence-a-local-data","title":"A Snapshot of Influence: A Local Data Attribution Framework for Online Reinforcement Learning","date":"2025-05-25","arxiv_id":"2505.19281","repositories_listed":1,"syntology":null},{"url":"/paper/satori-r1-incentivizing-multimodal-reasoning","slug":"satori-r1-incentivizing-multimodal-reasoning","title":"SATORI-R1: Incentivizing Multimodal Reasoning with Spatial Grounding and Verifiable Rewards","date":"2025-05-25","arxiv_id":"2505.19094","repositories_listed":1,"syntology":null},{"url":"/paper/serl-self-play-reinforcement-learning-for","slug":"serl-self-play-reinforcement-learning-for","title":"SeRL: Self-Play Reinforcement Learning for Large Language Models with Limited Data","date":"2025-05-25","arxiv_id":"2505.20347","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/serl-self-play-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2505.20347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20347"}},"official":{"repos":["wantbook-book/serl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/step-level-reward-for-free-in-rl-based-t2i","slug":"step-level-reward-for-free-in-rl-based-t2i","title":"Step-level Reward for Free in RL-based T2I Diffusion Model Fine-tuning","date":"2025-05-25","arxiv_id":"2505.19196","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/step-level-reward-for-free-in-rl-based-t2i#ran","syntology_url":"https://syntology.ai/paper/2505.19196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19196"}},"official":{"repos":["lil-shake/coca"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/structured-reinforcement-learning-for-1","slug":"structured-reinforcement-learning-for-1","title":"Structured Reinforcement Learning for Combinatorial Decision-Making","date":"2025-05-25","arxiv_id":"2505.19053","repositories_listed":1,"syntology":null},{"url":"/paper/veripo-cultivating-long-reasoning-in-video","slug":"veripo-cultivating-long-reasoning-in-video","title":"VerIPO: Cultivating Long Reasoning in Video-LLMs via Verifier-Gudied Iterative Policy Optimization","date":"2025-05-25","arxiv_id":"2505.19000","repositories_listed":1,"syntology":null},{"url":"/paper/adactrl-towards-adaptive-and-controllable","slug":"adactrl-towards-adaptive-and-controllable","title":"AdaCtrl: Towards Adaptive and Controllable Reasoning via Difficulty-Aware Budgeting","date":"2025-05-24","arxiv_id":"2505.18822","repositories_listed":1,"syntology":{"n":15,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/adactrl-towards-adaptive-and-controllable#ran","syntology_url":"https://syntology.ai/paper/2505.18822","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18822"}},"official":{"repos":["joeying1019/adactrl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-blend-inference-time-multi","slug":"diffusion-blend-inference-time-multi","title":"Diffusion Blend: Inference-Time Multi-Preference Alignment for Diffusion Models","date":"2025-05-24","arxiv_id":"2505.18547","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffusion-blend-inference-time-multi#ran","syntology_url":"https://syntology.ai/paper/2505.18547","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18547"}},"official":{"repos":["bluewoods127/db-2025"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-efficiency-and-exploration-in","slug":"enhancing-efficiency-and-exploration-in","title":"Enhancing Efficiency and Exploration in Reinforcement Learning for LLMs","date":"2025-05-24","arxiv_id":"2505.18573","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-latent-reasoning-via-reinforcement","slug":"hybrid-latent-reasoning-via-reinforcement","title":"Hybrid Latent Reasoning via Reinforcement Learning","date":"2025-05-24","arxiv_id":"2505.18454","repositories_listed":1,"syntology":null},{"url":"/paper/vla-rl-towards-masterful-and-general-robotic","slug":"vla-rl-towards-masterful-and-general-robotic","title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","date":"2025-05-24","arxiv_id":"2505.18719","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vla-rl-towards-masterful-and-general-robotic#ran","syntology_url":"https://syntology.ai/paper/2505.18719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18719"}},"official":{"repos":["guanxinglu/vlarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/co-reinforcement-learning-for-unified","slug":"co-reinforcement-learning-for-unified","title":"Co-Reinforcement Learning for Unified Multimodal Understanding and Generation","date":"2025-05-23","arxiv_id":"2505.17534","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-reinforcement-learning-for-unified#ran","syntology_url":"https://syntology.ai/paper/2505.17534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17534"}},"official":{"repos":["mm-vl/ulm-r1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/qwenlong-l1-towards-long-context-large","slug":"qwenlong-l1-towards-long-context-large","title":"QwenLong-L1: Towards Long-Context Large Reasoning Models with Reinforcement Learning","date":"2025-05-23","arxiv_id":"2505.17667","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-ballbot-navigation","slug":"reinforcement-learning-for-ballbot-navigation","title":"Reinforcement Learning for Ballbot Navigation in Uneven Terrain","date":"2025-05-23","arxiv_id":"2505.18417","repositories_listed":1,"syntology":null},{"url":"/paper/the-cell-must-go-on-agar-io-for-continual","slug":"the-cell-must-go-on-agar-io-for-continual","title":"The Cell Must Go On: Agar.io for Continual Reinforcement Learning","date":"2025-05-23","arxiv_id":"2505.18347","repositories_listed":1,"syntology":null},{"url":"/paper/thinking-fast-and-right-balancing-accuracy","slug":"thinking-fast-and-right-balancing-accuracy","title":"Thinking Fast and Right: Balancing Accuracy and Reasoning Length with Adaptive Rewards","date":"2025-05-23","arxiv_id":"2505.18298","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/thinking-fast-and-right-balancing-accuracy#ran","syntology_url":"https://syntology.ai/paper/2505.18298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18298"}},"official":{"repos":["jinyansu1/a-dlp"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/towards-revealing-the-effectiveness-of-small","slug":"towards-revealing-the-effectiveness-of-small","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","date":"2025-05-23","arxiv_id":"2505.17988","repositories_listed":1,"syntology":null},{"url":"/paper/wingpt-3-0-technical-report","slug":"wingpt-3-0-technical-report","title":"WiNGPT-3.0 Technical Report","date":"2025-05-23","arxiv_id":"2505.17387","repositories_listed":1,"syntology":null},{"url":"/paper/arctic-text2sql-r1-simple-rewards-strong","slug":"arctic-text2sql-r1-simple-rewards-strong","title":"Arctic-Text2SQL-R1: Simple Rewards, Strong Reasoning in Text-to-SQL","date":"2025-05-22","arxiv_id":"2505.20315","repositories_listed":1,"syntology":null},{"url":"/paper/arpo-end-to-end-policy-optimization-for-gui","slug":"arpo-end-to-end-policy-optimization-for-gui","title":"ARPO:End-to-End Policy Optimization for GUI Agents with Experience Replay","date":"2025-05-22","arxiv_id":"2505.16282","repositories_listed":1,"syntology":{"n":16,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/arpo-end-to-end-policy-optimization-for-gui#ran","syntology_url":"https://syntology.ai/paper/2505.16282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16282"}},"official":{"repos":["dvlab-research/arpo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/pytupli-a-scalable-infrastructure-for","slug":"pytupli-a-scalable-infrastructure-for","title":"PyTupli: A Scalable Infrastructure for Collaborative Offline Reinforcement Learning Projects","date":"2025-05-22","arxiv_id":"2505.16754","repositories_listed":1,"syntology":null},{"url":"/paper/saturn-sat-based-reinforcement-learning-to","slug":"saturn-sat-based-reinforcement-learning-to","title":"SATURN: SAT-based Reinforcement Learning to Unleash Language Model Reasoning","date":"2025-05-22","arxiv_id":"2505.16368","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/saturn-sat-based-reinforcement-learning-to#ran","syntology_url":"https://syntology.ai/paper/2505.16368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16368"}},"official":{"repos":["gtxygyzb/saturn-code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"url":"/paper/ssr-zero-simple-self-rewarding-reinforcement","slug":"ssr-zero-simple-self-rewarding-reinforcement","title":"SSR-Zero: Simple Self-Rewarding Reinforcement Learning for Machine Translation","date":"2025-05-22","arxiv_id":"2505.16637","repositories_listed":1,"syntology":null},{"url":"/paper/swe-dev-evaluating-and-training-autonomous","slug":"swe-dev-evaluating-and-training-autonomous","title":"SWE-Dev: Evaluating and Training Autonomous Feature-Driven Software Development","date":"2025-05-22","arxiv_id":"2505.16975","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/swe-dev-evaluating-and-training-autonomous#ran","syntology_url":"https://syntology.ai/paper/2505.16975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16975"}},"official":{"repos":["dorothyduuu/swe-dev"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/think-or-not-selective-reasoning-via","slug":"think-or-not-selective-reasoning-via","title":"Think or Not? Selective Reasoning via Reinforcement Learning for Vision-Language Models","date":"2025-05-22","arxiv_id":"2505.16854","repositories_listed":1,"syntology":null},{"url":"/paper/think-rm-enabling-long-horizon-reasoning-in","slug":"think-rm-enabling-long-horizon-reasoning-in","title":"Think-RM: Enabling Long-Horizon Reasoning in Generative Reward Models","date":"2025-05-22","arxiv_id":"2505.16265","repositories_listed":1,"syntology":null},{"url":"/paper/tool-star-empowering-llm-brained-multi-tool","slug":"tool-star-empowering-llm-brained-multi-tool","title":"Tool-Star: Empowering LLM-Brained Multi-Tool Reasoner via Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.16410","repositories_listed":1,"syntology":null},{"url":"/paper/webagent-r1-training-web-agents-via-end-to","slug":"webagent-r1-training-web-agents-via-end-to","title":"WebAgent-R1: Training Web Agents via End-to-End Multi-Turn Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.16421","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/webagent-r1-training-web-agents-via-end-to#ran","syntology_url":"https://syntology.ai/paper/2505.16421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16421"}},"official":{"repos":["weizhepei/webagent-r1"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-temporal-difference-method-for-stochastic","slug":"a-temporal-difference-method-for-stochastic","title":"A Temporal Difference Method for Stochastic Continuous Dynamics","date":"2025-05-21","arxiv_id":"2505.15544","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-on-reinforcement-learning","slug":"an-empirical-study-on-reinforcement-learning","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","date":"2025-05-21","arxiv_id":"2505.15117","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-on-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2505.15117","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15117"}},"official":{"repos":["petergriffinjin/search-r1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/from-problem-solving-to-teaching-problem","slug":"from-problem-solving-to-teaching-problem","title":"From Problem-Solving to Teaching Problem-Solving: Aligning LLMs with Pedagogy using Reinforcement Learning","date":"2025-05-21","arxiv_id":"2505.15607","repositories_listed":1,"syntology":null},{"url":"/paper/gui-g1-understanding-r1-zero-like-training","slug":"gui-g1-understanding-r1-zero-like-training","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","date":"2025-05-21","arxiv_id":"2505.15810","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/gui-g1-understanding-r1-zero-like-training#ran","syntology_url":"https://syntology.ai/paper/2505.15810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15810"}},"official":{"repos":["yuqi-zhou/gui-g1"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/guided-policy-optimization-under-partial","slug":"guided-policy-optimization-under-partial","title":"Guided Policy Optimization under Partial Observability","date":"2025-05-21","arxiv_id":"2505.15418","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/guided-policy-optimization-under-partial#ran","syntology_url":"https://syntology.ai/paper/2505.15418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15418"}},"official":{"repos":["liyheng/GPO"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/learn-to-reason-efficiently-with-adaptive","slug":"learn-to-reason-efficiently-with-adaptive","title":"Learn to Reason Efficiently with Adaptive Length-based Reward Shaping","date":"2025-05-21","arxiv_id":"2505.15612","repositories_listed":1,"syntology":null},{"url":"/paper/mmada-multimodal-large-diffusion-language","slug":"mmada-multimodal-large-diffusion-language","title":"MMaDA: Multimodal Large Diffusion Language Models","date":"2025-05-21","arxiv_id":"2505.15809","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mmada-multimodal-large-diffusion-language#ran","syntology_url":"https://syntology.ai/paper/2505.15809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15809"}},"official":{"repos":["gen-verse/mmada"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rl-tango-reinforcing-generator-and-verifier","slug":"rl-tango-reinforcing-generator-and-verifier","title":"RL Tango: Reinforcing Generator and Verifier Together for Language Reasoning","date":"2025-05-21","arxiv_id":"2505.15034","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rl-tango-reinforcing-generator-and-verifier#ran","syntology_url":"https://syntology.ai/paper/2505.15034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15034"}},"official":{"repos":["kaiwenzha/rl-tango"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rlbenchnet-the-right-network-for-the-right","slug":"rlbenchnet-the-right-network-for-the-right","title":"RLBenchNet: The Right Network for the Right Reinforcement Learning Task","date":"2025-05-21","arxiv_id":"2505.15040","repositories_listed":1,"syntology":null},{"url":"/paper/apex-empowering-llms-with-physics-based-task","slug":"apex-empowering-llms-with-physics-based-task","title":"APEX: Empowering LLMs with Physics-Based Task Planning for Real-time Insight","date":"2025-05-20","arxiv_id":"2505.13921","repositories_listed":1,"syntology":null},{"url":"/paper/general-reasoner-advancing-llm-reasoning","slug":"general-reasoner-advancing-llm-reasoning","title":"General-Reasoner: Advancing LLM Reasoning Across All Domains","date":"2025-05-20","arxiv_id":"2505.14652","repositories_listed":1,"syntology":null},{"url":"/paper/s3-you-don-t-need-that-much-data-to-train-a","slug":"s3-you-don-t-need-that-much-data-to-train-a","title":"s3: You Don't Need That Much Data to Train a Search Agent via RL","date":"2025-05-20","arxiv_id":"2505.14146","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":3,"n_ran_checked":3,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/s3-you-don-t-need-that-much-data-to-train-a#ran","syntology_url":"https://syntology.ai/paper/2505.14146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14146"}},"official":{"repos":["pat-jj/s3"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/think-j-learning-to-think-for-generative-llm","slug":"think-j-learning-to-think-for-generative-llm","title":"Think-J: Learning to Think for Generative LLM-as-a-Judge","date":"2025-05-20","arxiv_id":"2505.14268","repositories_listed":1,"syntology":null},{"url":"/paper/tinyv-reducing-false-negatives-in","slug":"tinyv-reducing-false-negatives-in","title":"TinyV: Reducing False Negatives in Verification Improves RL for LLM Reasoning","date":"2025-05-20","arxiv_id":"2505.14625","repositories_listed":1,"syntology":{"n":19,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/tinyv-reducing-false-negatives-in#ran","syntology_url":"https://syntology.ai/paper/2505.14625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14625"}},"official":{"repos":["uw-nsl/tinyv"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-moeas-for-solving-continuous","slug":"benchmarking-moeas-for-solving-continuous","title":"Benchmarking MOEAs for solving continuous multi-objective RL problems","date":"2025-05-19","arxiv_id":"2505.13726","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-explanations-for-continuous","slug":"counterfactual-explanations-for-continuous","title":"Counterfactual Explanations for Continuous Action Reinforcement Learning","date":"2025-05-19","arxiv_id":"2505.12701","repositories_listed":1,"syntology":null},{"url":"/paper/do-not-let-low-probability-tokens-over","slug":"do-not-let-low-probability-tokens-over","title":"Do Not Let Low-Probability Tokens Over-Dominate in RL for LLMs","date":"2025-05-19","arxiv_id":"2505.12929","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/do-not-let-low-probability-tokens-over#ran","syntology_url":"https://syntology.ai/paper/2505.12929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12929"}},"official":{"repos":["zhyang2226/ar-lopti"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/effective-and-transparent-rag-adaptive-reward","slug":"effective-and-transparent-rag-adaptive-reward","title":"Effective and Transparent RAG: Adaptive-Reward Reinforcement Learning for Decision Traceability","date":"2025-05-19","arxiv_id":"2505.13258","repositories_listed":1,"syntology":null}],"record_sha256":"6b0f8ad5e35600920db2f1f167803a68c08bdb532c2fff05ba764d7b35b6a44b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}