{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/9","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":135,"rows_per_page":100,"rows":[801,900],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/8","next":"/task/reinforcement-learning-2/papers/10","papers":[{"url":"/paper/learning-and-policy-search-in-stochastic","slug":"learning-and-policy-search-in-stochastic","title":"Learning and Policy Search in Stochastic Dynamical Systems with Bayesian Neural Networks","date":"2016-05-23","arxiv_id":"1605.07127","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-large-discrete","slug":"deep-reinforcement-learning-in-large-discrete","title":"Deep Reinforcement Learning in Large Discrete Action Spaces","date":"2015-12-24","arxiv_id":"1512.07679","repositories_listed":2,"syntology":null},{"url":"/paper/increasing-the-action-gap-new-operators-for","slug":"increasing-the-action-gap-new-operators-for","title":"Increasing the Action Gap: New Operators for Reinforcement Learning","date":"2015-12-15","arxiv_id":"1512.04860","repositories_listed":2,"syntology":null},{"url":"/paper/doubly-robust-off-policy-value-evaluation-for","slug":"doubly-robust-off-policy-value-evaluation-for","title":"Doubly Robust Off-policy Value Evaluation for Reinforcement Learning","date":"2015-11-11","arxiv_id":"1511.03722","repositories_listed":2,"syntology":null},{"url":"/paper/variational-information-maximisation-for","slug":"variational-information-maximisation-for","title":"Variational Information Maximisation for Intrinsically Motivated Reinforcement Learning","date":"2015-09-29","arxiv_id":"1509.08731","repositories_listed":2,"syntology":null},{"url":"/paper/maximum-entropy-deep-inverse-reinforcement","slug":"maximum-entropy-deep-inverse-reinforcement","title":"Maximum Entropy Deep Inverse Reinforcement Learning","date":"2015-07-17","arxiv_id":"1507.04888","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/maximum-entropy-deep-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1507.04888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1507.04888"}},"official":null}},{"url":"/paper/a-monte-carlo-aixi-approximation","slug":"a-monte-carlo-aixi-approximation","title":"A Monte Carlo AIXI Approximation","date":"2009-09-04","arxiv_id":"0909.0801","repositories_listed":2,"syntology":null},{"url":"/paper/quantum-reinforcement-learning","slug":"quantum-reinforcement-learning","title":"Quantum reinforcement learning","date":"2008-10-21","arxiv_id":"0810.3828","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantum-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/0810.3828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"0810.3828"}},"official":null}},{"url":"/paper/exploring-the-robustness-of-tractoracle","slug":"exploring-the-robustness-of-tractoracle","title":"Exploring the robustness of TractOracle methods in RL-based tractography","date":"2025-07-15","arxiv_id":"2507.11486","repositories_listed":1,"syntology":null},{"url":"/paper/high-throughput-distributed-reinforcement","slug":"high-throughput-distributed-reinforcement","title":"High-Throughput Distributed Reinforcement Learning via Adaptive Policy Synchronization","date":"2025-07-15","arxiv_id":"2507.10990","repositories_listed":1,"syntology":null},{"url":"/paper/step-wise-policy-for-rare-tool-knowledge","slug":"step-wise-policy-for-rare-tool-knowledge","title":"Step-wise Policy for Rare-tool Knowledge (SPaRK): Offline RL that Drives Diverse Tool Use in LLMs","date":"2025-07-15","arxiv_id":"2507.11371","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-gradient-1","slug":"deep-reinforcement-learning-with-gradient-1","title":"Deep Reinforcement Learning with Gradient Eligibility Traces","date":"2025-07-12","arxiv_id":"2507.09087","repositories_listed":1,"syntology":null},{"url":"/paper/autotriton-automatic-triton-programming-with","slug":"autotriton-automatic-triton-programming-with","title":"AutoTriton: Automatic Triton Programming with Reinforcement Learning in LLMs","date":"2025-07-08","arxiv_id":"2507.05687","repositories_listed":1,"syntology":null},{"url":"/paper/criticlean-critic-guided-reinforcement","slug":"criticlean-critic-guided-reinforcement","title":"CriticLean: Critic-Guided Reinforcement Learning for Mathematical Formalization","date":"2025-07-08","arxiv_id":"2507.06181","repositories_listed":1,"syntology":null},{"url":"/paper/rlver-reinforcement-learning-with-verifiable","slug":"rlver-reinforcement-learning-with-verifiable","title":"RLVER: Reinforcement Learning with Verifiable Emotion Rewards for Empathetic Agents","date":"2025-07-03","arxiv_id":"2507.03112","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rlver-reinforcement-learning-with-verifiable#ran","syntology_url":"https://syntology.ai/paper/2507.03112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.03112"}},"official":{"repos":["tencent/digitalhuman"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/l0-reinforcement-learning-to-become-general-1","slug":"l0-reinforcement-learning-to-become-general-1","title":"L0: Reinforcement Learning to Become General Agents","date":"2025-06-30","arxiv_id":"2506.23667","repositories_listed":1,"syntology":null},{"url":"/paper/octothinker-mid-training-incentivizes","slug":"octothinker-mid-training-incentivizes","title":"OctoThinker: Mid-training Incentivizes Reinforcement Learning Scaling","date":"2025-06-25","arxiv_id":"2506.20512","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":4,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/octothinker-mid-training-incentivizes#ran","syntology_url":"https://syntology.ai/paper/2506.20512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20512"}},"official":{"repos":["gair-nlp/octothinker"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/recode-updating-code-api-knowledge-with","slug":"recode-updating-code-api-knowledge-with","title":"ReCode: Updating Code API Knowledge with Reinforcement Learning","date":"2025-06-25","arxiv_id":"2506.20495","repositories_listed":1,"syntology":null},{"url":"/paper/knowrl-exploring-knowledgeable-reinforcement","slug":"knowrl-exploring-knowledgeable-reinforcement","title":"KnowRL: Exploring Knowledgeable Reinforcement Learning for Factuality","date":"2025-06-24","arxiv_id":"2506.19807","repositories_listed":1,"syntology":null},{"url":"/paper/multi-preference-lambda-weighted-listwise-dpo","slug":"multi-preference-lambda-weighted-listwise-dpo","title":"Multi-Preference Lambda-weighted Listwise DPO for Dynamic Preference Alignment","date":"2025-06-24","arxiv_id":"2506.19780","repositories_listed":1,"syntology":null},{"url":"/paper/partially-observable-residual-reinforcement","slug":"partially-observable-residual-reinforcement","title":"Partially Observable Residual Reinforcement Learning for PV-Inverter-Based Voltage Control in Distribution Grids","date":"2025-06-24","arxiv_id":"2506.19353","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-reg-improving-sample-complexity-in","slug":"sparse-reg-improving-sample-complexity-in","title":"Sparse-Reg: Improving Sample Complexity in Offline Reinforcement Learning using Sparsity","date":"2025-06-20","arxiv_id":"2506.17155","repositories_listed":1,"syntology":null},{"url":"/paper/transdreamerv3-implanting-transformer-in","slug":"transdreamerv3-implanting-transformer-in","title":"TransDreamerV3: Implanting Transformer In DreamerV3","date":"2025-06-20","arxiv_id":"2506.17103","repositories_listed":1,"syntology":null},{"url":"/paper/dr-sac-distributionally-robust-soft-actor","slug":"dr-sac-distributionally-robust-soft-actor","title":"DR-SAC: Distributionally Robust Soft Actor-Critic for Reinforcement Learning under Uncertainty","date":"2025-06-14","arxiv_id":"2506.12622","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/dr-sac-distributionally-robust-soft-actor#ran","syntology_url":"https://syntology.ai/paper/2506.12622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12622"}},"official":{"repos":["lemutisme/dr-sac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/schema-r1-a-reasoning-training-approach-for","slug":"schema-r1-a-reasoning-training-approach-for","title":"Schema-R1: A reasoning training approach for schema linking in Text-to-SQL Task","date":"2025-06-13","arxiv_id":"2506.11986","repositories_listed":1,"syntology":null},{"url":"/paper/treerl-llm-reinforcement-learning-with-on","slug":"treerl-llm-reinforcement-learning-with-on","title":"TreeRL: LLM Reinforcement Learning with On-Policy Tree Search","date":"2025-06-13","arxiv_id":"2506.11902","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/treerl-llm-reinforcement-learning-with-on#ran","syntology_url":"https://syntology.ai/paper/2506.11902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.11902"}},"official":{"repos":["thudm/treerl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-pre-training-on-unlabeled-images-using","slug":"visual-pre-training-on-unlabeled-images-using","title":"Visual Pre-Training on Unlabeled Images using Reinforcement Learning","date":"2025-06-13","arxiv_id":"2506.11967","repositories_listed":1,"syntology":null},{"url":"/paper/verif-verification-engineering-for","slug":"verif-verification-engineering-for","title":"VerIF: Verification Engineering for Reinforcement Learning in Instruction Following","date":"2025-06-11","arxiv_id":"2506.09942","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/verif-verification-engineering-for#ran","syntology_url":"https://syntology.ai/paper/2506.09942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09942"}},"official":{"repos":["thu-keg/verif"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2506-08460","slug":"2506-08460","title":"MOBODY: Model Based Off-Dynamics Offline Reinforcement Learning","date":"2025-06-10","arxiv_id":"2506.08460","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2506-08460#ran","syntology_url":"https://syntology.ai/paper/2506.08460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08460"}},"official":{"repos":["guoyihonggyh/mobody-model-based-off-dynamics-offline-reinforcement-learning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/agile-reinforcement-learning-for-real-time","slug":"agile-reinforcement-learning-for-real-time","title":"Agile Reinforcement Learning for Real-Time Task Scheduling in Edge Computing","date":"2025-06-10","arxiv_id":"2506.08850","repositories_listed":1,"syntology":null},{"url":"/paper/consistent-paths-lead-to-truth-self-rewarding","slug":"consistent-paths-lead-to-truth-self-rewarding","title":"Consistent Paths Lead to Truth: Self-Rewarding Reinforcement Learning for LLM Reasoning","date":"2025-06-10","arxiv_id":"2506.08745","repositories_listed":1,"syntology":null},{"url":"/paper/logicpuzzlerl-cultivating-robust-mathematical","slug":"logicpuzzlerl-cultivating-robust-mathematical","title":"LogicPuzzleRL: Cultivating Robust Mathematical Reasoning in LLMs via Reinforcement Learning","date":"2025-06-05","arxiv_id":"2506.04821","repositories_listed":1,"syntology":null},{"url":"/paper/egovlm-policy-optimization-for-egocentric","slug":"egovlm-policy-optimization-for-egocentric","title":"EgoVLM: Policy Optimization for Egocentric Video Understanding","date":"2025-06-03","arxiv_id":"2506.03097","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-table-exploring-reinforcement","slug":"reasoning-table-exploring-reinforcement","title":"Reasoning-Table: Exploring Reinforcement Learning for Table Reasoning","date":"2025-06-02","arxiv_id":"2506.01710","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-tuning-for-videollms","slug":"reinforcement-learning-tuning-for-videollms","title":"Reinforcement Learning Tuning for VideoLLMs: Reward Design and Data Efficiency","date":"2025-06-02","arxiv_id":"2506.01908","repositories_listed":1,"syntology":{"n":17,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/reinforcement-learning-tuning-for-videollms#ran","syntology_url":"https://syntology.ai/paper/2506.01908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01908"}},"official":{"repos":["appletea233/temporal-r1"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/infinity-parser-layout-aware-reinforcement","slug":"infinity-parser-layout-aware-reinforcement","title":"Infinity Parser: Layout Aware Reinforcement Learning for Scanned Document Parsing","date":"2025-06-01","arxiv_id":"2506.03197","repositories_listed":1,"syntology":null},{"url":"/paper/clarify-contrastive-preference-reinforcement","slug":"clarify-contrastive-preference-reinforcement","title":"CLARIFY: Contrastive Preference Reinforcement Learning for Untangling Ambiguous Queries","date":"2025-05-31","arxiv_id":"2506.00388","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":1,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":6,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clarify-contrastive-preference-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2506.00388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00388"}},"official":{"repos":["moonoutcloudback/clarify_pbrl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/biological-pathway-guided-gene-selection","slug":"biological-pathway-guided-gene-selection","title":"Biological Pathway Guided Gene Selection Through Collaborative Reinforcement Learning","date":"2025-05-30","arxiv_id":"2505.24155","repositories_listed":1,"syntology":null},{"url":"/paper/mofgpt-generative-design-of-metal-organic","slug":"mofgpt-generative-design-of-metal-organic","title":"MOFGPT: Generative Design of Metal-Organic Frameworks using Language Models","date":"2025-05-30","arxiv_id":"2506.00198","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-gym-reasoning-environments-for","slug":"reasoning-gym-reasoning-environments-for","title":"REASONING GYM: Reasoning Environments for Reinforcement Learning with Verifiable Rewards","date":"2025-05-30","arxiv_id":"2505.24760","repositories_listed":1,"syntology":null},{"url":"/paper/composite-reward-design-in-ppo-driven","slug":"composite-reward-design-in-ppo-driven","title":"Composite Reward Design in PPO-Driven Adaptive Filtering","date":"2025-05-29","arxiv_id":"2506.06323","repositories_listed":1,"syntology":null},{"url":"/paper/grounded-reinforcement-learning-for-visual","slug":"grounded-reinforcement-learning-for-visual","title":"Grounded Reinforcement Learning for Visual Reasoning","date":"2025-05-29","arxiv_id":"2505.23678","repositories_listed":1,"syntology":null},{"url":"/paper/on-policy-rl-with-optimal-reward-baseline","slug":"on-policy-rl-with-optimal-reward-baseline","title":"On-Policy RL with Optimal Reward Baseline","date":"2025-05-29","arxiv_id":"2505.23585","repositories_listed":1,"syntology":null},{"url":"/paper/towards-reward-fairness-in-rlhf-from-a","slug":"towards-reward-fairness-in-rlhf-from-a","title":"Towards Reward Fairness in RLHF: From a Resource Allocation Perspective","date":"2025-05-29","arxiv_id":"2505.23349","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-reward-fairness-in-rlhf-from-a#ran","syntology_url":"https://syntology.ai/paper/2505.23349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23349"}},"official":{"repos":["shoyua/towards-reward-fairness"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cadrille-multi-modal-cad-reconstruction-with","slug":"cadrille-multi-modal-cad-reconstruction-with","title":"cadrille: Multi-modal CAD Reconstruction with Online Reinforcement Learning","date":"2025-05-28","arxiv_id":"2505.22914","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/cadrille-multi-modal-cad-reconstruction-with#ran","syntology_url":"https://syntology.ai/paper/2505.22914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22914"}},"official":null}},{"url":"/paper/sorel-and-torel-two-methods-for-fully-offline","slug":"sorel-and-torel-two-methods-for-fully-offline","title":"SOReL and TOReL: Two Methods for Fully Offline Reinforcement Learning","date":"2025-05-28","arxiv_id":"2505.22442","repositories_listed":1,"syntology":null},{"url":"/paper/when-does-neuroevolution-outcompete","slug":"when-does-neuroevolution-outcompete","title":"When Does Neuroevolution Outcompete Reinforcement Learning in Transfer Learning Tasks?","date":"2025-05-28","arxiv_id":"2505.22696","repositories_listed":1,"syntology":null},{"url":"/paper/discover-automated-curricula-for-sparse","slug":"discover-automated-curricula-for-sparse","title":"DISCOVER: Automated Curricula for Sparse-Reward Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.19850","repositories_listed":1,"syntology":null},{"url":"/paper/incentivizing-reasoning-from-weak-supervision","slug":"incentivizing-reasoning-from-weak-supervision","title":"Incentivizing Reasoning from Weak Supervision","date":"2025-05-26","arxiv_id":"2505.20072","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-reason-without-external-rewards","slug":"learning-to-reason-without-external-rewards","title":"Learning to Reason without External Rewards","date":"2025-05-26","arxiv_id":"2505.19590","repositories_listed":1,"syntology":null},{"url":"/paper/one-shot-entropy-minimization","slug":"one-shot-entropy-minimization","title":"One-shot Entropy Minimization","date":"2025-05-26","arxiv_id":"2505.20282","repositories_listed":1,"syntology":null},{"url":"/paper/rearank-reasoning-re-ranking-agent-via","slug":"rearank-reasoning-re-ranking-agent-via","title":"REARANK: Reasoning Re-ranking Agent via Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.20046","repositories_listed":1,"syntology":{"n":18,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/rearank-reasoning-re-ranking-agent-via#ran","syntology_url":"https://syntology.ai/paper/2505.20046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20046"}},"official":{"repos":["lezhang7/rearank"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/refining-few-step-text-to-multiview-diffusion","slug":"refining-few-step-text-to-multiview-diffusion","title":"Refining Few-Step Text-to-Multiview Diffusion via Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.20107","repositories_listed":1,"syntology":null},{"url":"/paper/behavior-injection-preparing-language-models","slug":"behavior-injection-preparing-language-models","title":"Behavior Injection: Preparing Language Models for Reinforcement Learning","date":"2025-05-25","arxiv_id":"2505.18917","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/behavior-injection-preparing-language-models#ran","syntology_url":"https://syntology.ai/paper/2505.18917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18917"}},"official":{"repos":["czp16/bridge-llm-reasoning"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/serl-self-play-reinforcement-learning-for","slug":"serl-self-play-reinforcement-learning-for","title":"SeRL: Self-Play Reinforcement Learning for Large Language Models with Limited Data","date":"2025-05-25","arxiv_id":"2505.20347","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/serl-self-play-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2505.20347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20347"}},"official":{"repos":["wantbook-book/serl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/structured-reinforcement-learning-for-1","slug":"structured-reinforcement-learning-for-1","title":"Structured Reinforcement Learning for Combinatorial Decision-Making","date":"2025-05-25","arxiv_id":"2505.19053","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-meta-reinforcement-learning-with","slug":"bayesian-meta-reinforcement-learning-with","title":"Bayesian Meta-Reinforcement Learning with Laplace Variational Recurrent Networks","date":"2025-05-24","arxiv_id":"2505.18591","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-efficiency-and-exploration-in","slug":"enhancing-efficiency-and-exploration-in","title":"Enhancing Efficiency and Exploration in Reinforcement Learning for LLMs","date":"2025-05-24","arxiv_id":"2505.18573","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-latent-reasoning-via-reinforcement","slug":"hybrid-latent-reasoning-via-reinforcement","title":"Hybrid Latent Reasoning via Reinforcement Learning","date":"2025-05-24","arxiv_id":"2505.18454","repositories_listed":1,"syntology":null},{"url":"/paper/a-robust-ppo-optimized-tabular-transformer","slug":"a-robust-ppo-optimized-tabular-transformer","title":"A Robust PPO-optimized Tabular Transformer Framework for Intrusion Detection in Industrial IoT Systems","date":"2025-05-23","arxiv_id":"2505.18234","repositories_listed":1,"syntology":null},{"url":"/paper/co-reinforcement-learning-for-unified","slug":"co-reinforcement-learning-for-unified","title":"Co-Reinforcement Learning for Unified Multimodal Understanding and Generation","date":"2025-05-23","arxiv_id":"2505.17534","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-reinforcement-learning-for-unified#ran","syntology_url":"https://syntology.ai/paper/2505.17534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17534"}},"official":{"repos":["mm-vl/ulm-r1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/reinforcement-learning-for-ballbot-navigation","slug":"reinforcement-learning-for-ballbot-navigation","title":"Reinforcement Learning for Ballbot Navigation in Uneven Terrain","date":"2025-05-23","arxiv_id":"2505.18417","repositories_listed":1,"syntology":null},{"url":"/paper/wingpt-3-0-technical-report","slug":"wingpt-3-0-technical-report","title":"WiNGPT-3.0 Technical Report","date":"2025-05-23","arxiv_id":"2505.17387","repositories_listed":1,"syntology":null},{"url":"/paper/arpo-end-to-end-policy-optimization-for-gui","slug":"arpo-end-to-end-policy-optimization-for-gui","title":"ARPO:End-to-End Policy Optimization for GUI Agents with Experience Replay","date":"2025-05-22","arxiv_id":"2505.16282","repositories_listed":1,"syntology":{"n":16,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/arpo-end-to-end-policy-optimization-for-gui#ran","syntology_url":"https://syntology.ai/paper/2505.16282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16282"}},"official":{"repos":["dvlab-research/arpo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/conciserl-conciseness-guided-reinforcement","slug":"conciserl-conciseness-guided-reinforcement","title":"ConciseRL: Conciseness-Guided Reinforcement Learning for Efficient Reasoning Models","date":"2025-05-22","arxiv_id":"2505.17250","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conciserl-conciseness-guided-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2505.17250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17250"}},"official":{"repos":["razvandu/conciserl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fact-r1-towards-explainable-video","slug":"fact-r1-towards-explainable-video","title":"Fact-R1: Towards Explainable Video Misinformation Detection with Deep Reasoning","date":"2025-05-22","arxiv_id":"2505.16836","repositories_listed":1,"syntology":null},{"url":"/paper/got-r1-unleashing-reasoning-capability-of","slug":"got-r1-unleashing-reasoning-capability-of","title":"GoT-R1: Unleashing Reasoning Capability of MLLM for Visual Generation with Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.17022","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":6,"n_ran_checked":7,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"9 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/got-r1-unleashing-reasoning-capability-of#ran","syntology_url":"https://syntology.ai/paper/2505.17022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17022"}},"official":{"repos":["gogoduan/got-r1"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ktae-a-model-free-algorithm-to-key-tokens","slug":"ktae-a-model-free-algorithm-to-key-tokens","title":"KTAE: A Model-Free Algorithm to Key-Tokens Advantage Estimation in Mathematical Reasoning","date":"2025-05-22","arxiv_id":"2505.16826","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/ktae-a-model-free-algorithm-to-key-tokens#ran","syntology_url":"https://syntology.ai/paper/2505.16826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16826"}},"official":{"repos":["xiaolizh1/ktae"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/maximum-total-correlation-reinforcement","slug":"maximum-total-correlation-reinforcement","title":"Maximum Total Correlation Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.16734","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":6,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"8 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/maximum-total-correlation-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2505.16734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16734"}},"official":{"repos":["bangyou01/mtc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-temporal-difference-method-for-stochastic","slug":"a-temporal-difference-method-for-stochastic","title":"A Temporal Difference Method for Stochastic Continuous Dynamics","date":"2025-05-21","arxiv_id":"2505.15544","repositories_listed":1,"syntology":null},{"url":"/paper/avatarshield-visual-reinforcement-learning","slug":"avatarshield-visual-reinforcement-learning","title":"AvatarShield: Visual Reinforcement Learning for Human-Centric Video Forgery Detection","date":"2025-05-21","arxiv_id":"2505.15173","repositories_listed":1,"syntology":null},{"url":"/paper/convsearch-r1-enhancing-query-reformulation","slug":"convsearch-r1-enhancing-query-reformulation","title":"ConvSearch-R1: Enhancing Query Reformulation for Conversational Search with Reasoning via Reinforcement Learning","date":"2025-05-21","arxiv_id":"2505.15776","repositories_listed":1,"syntology":null},{"url":"/paper/hadamax-encoding-elevating-performance-in","slug":"hadamax-encoding-elevating-performance-in","title":"Hadamax Encoding: Elevating Performance in Model-Free Atari","date":"2025-05-21","arxiv_id":"2505.15345","repositories_listed":1,"syntology":null},{"url":"/paper/nover-incentive-training-for-language-models","slug":"nover-incentive-training-for-language-models","title":"NOVER: Incentive Training for Language Models via Verifier-Free Reinforcement Learning","date":"2025-05-21","arxiv_id":"2505.16022","repositories_listed":1,"syntology":null},{"url":"/paper/deepeyes-incentivizing-thinking-with-images","slug":"deepeyes-incentivizing-thinking-with-images","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14362","repositories_listed":1,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deepeyes-incentivizing-thinking-with-images#ran","syntology_url":"https://syntology.ai/paper/2505.14362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14362"}},"official":{"repos":["visual-agent/deepeyes"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/korgym-a-dynamic-game-platform-for-llm","slug":"korgym-a-dynamic-game-platform-for-llm","title":"KORGym: A Dynamic Game Platform for LLM Reasoning Evaluation","date":"2025-05-20","arxiv_id":"2505.14552","repositories_listed":1,"syntology":null},{"url":"/paper/prl-prompts-from-reinforcement-learning","slug":"prl-prompts-from-reinforcement-learning","title":"PRL: Prompts from Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14412","repositories_listed":1,"syntology":null},{"url":"/paper/rlvr-world-training-world-models-with","slug":"rlvr-world-training-world-models-with","title":"RLVR-World: Training World Models with Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.13934","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rlvr-world-training-world-models-with#ran","syntology_url":"https://syntology.ai/paper/2505.13934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13934"}},"official":{"repos":["thuml/RLVR-World"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-and-computationally-efficient-1","slug":"sample-and-computationally-efficient-1","title":"Sample and Computationally Efficient Continuous-Time Reinforcement Learning with General Function Approximation","date":"2025-05-20","arxiv_id":"2505.14821","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-moeas-for-solving-continuous","slug":"benchmarking-moeas-for-solving-continuous","title":"Benchmarking MOEAs for solving continuous multi-objective RL problems","date":"2025-05-19","arxiv_id":"2505.13726","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-explanations-for-continuous","slug":"counterfactual-explanations-for-continuous","title":"Counterfactual Explanations for Continuous Action Reinforcement Learning","date":"2025-05-19","arxiv_id":"2505.12701","repositories_listed":1,"syntology":null},{"url":"/paper/dual-agent-reinforcement-learning-for","slug":"dual-agent-reinforcement-learning-for","title":"Dual-Agent Reinforcement Learning for Automated Feature Generation","date":"2025-05-19","arxiv_id":"2505.12628","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dual-agent-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2505.12628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12628"}},"official":{"repos":["extess0/darl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/extrans-multilingual-deep-reasoning","slug":"extrans-multilingual-deep-reasoning","title":"ExTrans: Multilingual Deep Reasoning Translation via Exemplar-Enhanced Reinforcement Learning","date":"2025-05-19","arxiv_id":"2505.12996","repositories_listed":1,"syntology":null},{"url":"/paper/retrospex-language-agent-meets-offline","slug":"retrospex-language-agent-meets-offline","title":"Retrospex: Language Agent Meets Offline Reinforcement Learning Critic","date":"2025-05-17","arxiv_id":"2505.11807","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11409","slug":"2505-11409","title":"Visual Planning: Let's Think Only with Images","date":"2025-05-16","arxiv_id":"2505.11409","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2505-11409#ran","syntology_url":"https://syntology.ai/paper/2505.11409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11409"}},"official":{"repos":["yix8/visualplanning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reasoning-on-a-budget-miniaturizing-deepseek","slug":"reasoning-on-a-budget-miniaturizing-deepseek","title":"Reasoning on a Budget: Miniaturizing DeepSeek R1 with SFT-GRPO Alignment for Instruction-Tuned LLMs","date":"2025-05-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/beyond-aha-toward-systematic-meta-abilities","slug":"beyond-aha-toward-systematic-meta-abilities","title":"Beyond 'Aha!': Toward Systematic Meta-Abilities Alignment in Large Reasoning Models","date":"2025-05-15","arxiv_id":"2505.10554","repositories_listed":1,"syntology":null},{"url":"/paper/imaginebench-evaluating-reinforcement","slug":"imaginebench-evaluating-reinforcement","title":"ImagineBench: Evaluating Reinforcement Learning with Large Language Model Rollouts","date":"2025-05-15","arxiv_id":"2505.10010","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-the-known-decision-making-with","slug":"beyond-the-known-decision-making-with","title":"Beyond the Known: Decision Making with Counterfactual Reasoning Decision Transformer","date":"2025-05-14","arxiv_id":"2505.09114","repositories_listed":1,"syntology":null},{"url":"/paper/dancegrpo-unleashing-grpo-on-visual","slug":"dancegrpo-unleashing-grpo-on-visual","title":"DanceGRPO: Unleashing GRPO on Visual Generation","date":"2025-05-12","arxiv_id":"2505.07818","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dancegrpo-unleashing-grpo-on-visual#ran","syntology_url":"https://syntology.ai/paper/2505.07818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07818"}},"official":null}},{"url":"/paper/mle-dojo-interactive-environments-for","slug":"mle-dojo-interactive-environments-for","title":"MLE-Dojo: Interactive Environments for Empowering LLM Agents in Machine Learning Engineering","date":"2025-05-12","arxiv_id":"2505.07782","repositories_listed":1,"syntology":null},{"url":"/paper/a-critical-assessment-of-reinforcement","slug":"a-critical-assessment-of-reinforcement","title":"A critical assessment of reinforcement learning methods for microswimmer navigation in complex flows","date":"2025-05-08","arxiv_id":"2505.05525","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-and-robust-dbscan-with-multi-agent","slug":"adaptive-and-robust-dbscan-with-multi-agent","title":"Adaptive and Robust DBSCAN with Multi-agent Reinforcement Learning","date":"2025-05-07","arxiv_id":"2505.04339","repositories_listed":1,"syntology":null},{"url":"/paper/echoink-r1-exploring-audio-visual-reasoning","slug":"echoink-r1-exploring-audio-visual-reasoning","title":"EchoInk-R1: Exploring Audio-Visual Reasoning in Multimodal LLMs via Reinforcement Learning","date":"2025-05-07","arxiv_id":"2505.04623","repositories_listed":1,"syntology":null},{"url":"/paper/rlministyler-light-weight-rl-style-agent-for","slug":"rlministyler-light-weight-rl-style-agent-for","title":"RLMiniStyler: Light-weight RL Style Agent for Arbitrary Sequential Neural Style Generation","date":"2025-05-07","arxiv_id":"2505.04424","repositories_listed":1,"syntology":null},{"url":"/paper/unraveling-the-rainbow-can-value-based","slug":"unraveling-the-rainbow-can-value-based","title":"Unraveling the Rainbow: can value-based methods schedule?","date":"2025-05-06","arxiv_id":"2505.03323","repositories_listed":1,"syntology":null},{"url":"/paper/a-generalised-and-adaptable-reinforcement","slug":"a-generalised-and-adaptable-reinforcement","title":"A Generalised and Adaptable Reinforcement Learning Stopping Method","date":"2025-05-03","arxiv_id":"2505.01907","repositories_listed":1,"syntology":null},{"url":"/paper/directly-forecasting-belief-for-reinforcement","slug":"directly-forecasting-belief-for-reinforcement","title":"Directly Forecasting Belief for Reinforcement Learning with Delays","date":"2025-05-01","arxiv_id":"2505.00546","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-27","slug":"multi-agent-reinforcement-learning-for-27","title":"Multi-Agent Reinforcement Learning for Resources Allocation Optimization: A Survey","date":"2025-04-29","arxiv_id":"2504.21048","repositories_listed":1,"syntology":null},{"url":"/paper/rulebook-bringing-co-routines-to","slug":"rulebook-bringing-co-routines-to","title":"Rulebook: bringing co-routines to reinforcement learning environments","date":"2025-04-28","arxiv_id":"2504.19625","repositories_listed":1,"syntology":null}],"record_sha256":"19760f120aef9e162dd8622601cf9e055268abbf7e23cf28ffa17b3a16a48ab5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}