{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/11","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":132,"rows_per_page":100,"rows":[1001,1100],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/10","next":"/task/reinforcement-learning/papers/12","papers":[{"url":"/paper/reasoning-gym-reasoning-environments-for","slug":"reasoning-gym-reasoning-environments-for","title":"REASONING GYM: Reasoning Environments for Reinforcement Learning with Verifiable Rewards","date":"2025-05-30","arxiv_id":"2505.24760","repositories_listed":1,"syntology":null},{"url":"/paper/composite-reward-design-in-ppo-driven","slug":"composite-reward-design-in-ppo-driven","title":"Composite Reward Design in PPO-Driven Adaptive Filtering","date":"2025-05-29","arxiv_id":"2506.06323","repositories_listed":1,"syntology":null},{"url":"/paper/grounded-reinforcement-learning-for-visual","slug":"grounded-reinforcement-learning-for-visual","title":"Grounded Reinforcement Learning for Visual Reasoning","date":"2025-05-29","arxiv_id":"2505.23678","repositories_listed":1,"syntology":null},{"url":"/paper/on-policy-rl-with-optimal-reward-baseline","slug":"on-policy-rl-with-optimal-reward-baseline","title":"On-Policy RL with Optimal Reward Baseline","date":"2025-05-29","arxiv_id":"2505.23585","repositories_listed":1,"syntology":null},{"url":"/paper/towards-reward-fairness-in-rlhf-from-a","slug":"towards-reward-fairness-in-rlhf-from-a","title":"Towards Reward Fairness in RLHF: From a Resource Allocation Perspective","date":"2025-05-29","arxiv_id":"2505.23349","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-reward-fairness-in-rlhf-from-a#ran","syntology_url":"https://syntology.ai/paper/2505.23349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23349"}},"official":{"repos":["shoyua/towards-reward-fairness"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cadrille-multi-modal-cad-reconstruction-with","slug":"cadrille-multi-modal-cad-reconstruction-with","title":"cadrille: Multi-modal CAD Reconstruction with Online Reinforcement Learning","date":"2025-05-28","arxiv_id":"2505.22914","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/cadrille-multi-modal-cad-reconstruction-with#ran","syntology_url":"https://syntology.ai/paper/2505.22914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22914"}},"official":null}},{"url":"/paper/sorel-and-torel-two-methods-for-fully-offline","slug":"sorel-and-torel-two-methods-for-fully-offline","title":"SOReL and TOReL: Two Methods for Fully Offline Reinforcement Learning","date":"2025-05-28","arxiv_id":"2505.22442","repositories_listed":1,"syntology":null},{"url":"/paper/when-does-neuroevolution-outcompete","slug":"when-does-neuroevolution-outcompete","title":"When Does Neuroevolution Outcompete Reinforcement Learning in Transfer Learning Tasks?","date":"2025-05-28","arxiv_id":"2505.22696","repositories_listed":1,"syntology":null},{"url":"/paper/discover-automated-curricula-for-sparse","slug":"discover-automated-curricula-for-sparse","title":"DISCOVER: Automated Curricula for Sparse-Reward Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.19850","repositories_listed":1,"syntology":null},{"url":"/paper/incentivizing-reasoning-from-weak-supervision","slug":"incentivizing-reasoning-from-weak-supervision","title":"Incentivizing Reasoning from Weak Supervision","date":"2025-05-26","arxiv_id":"2505.20072","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-reason-without-external-rewards","slug":"learning-to-reason-without-external-rewards","title":"Learning to Reason without External Rewards","date":"2025-05-26","arxiv_id":"2505.19590","repositories_listed":1,"syntology":null},{"url":"/paper/one-shot-entropy-minimization","slug":"one-shot-entropy-minimization","title":"One-shot Entropy Minimization","date":"2025-05-26","arxiv_id":"2505.20282","repositories_listed":1,"syntology":null},{"url":"/paper/rearank-reasoning-re-ranking-agent-via","slug":"rearank-reasoning-re-ranking-agent-via","title":"REARANK: Reasoning Re-ranking Agent via Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.20046","repositories_listed":1,"syntology":{"n":18,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/rearank-reasoning-re-ranking-agent-via#ran","syntology_url":"https://syntology.ai/paper/2505.20046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20046"}},"official":{"repos":["lezhang7/rearank"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/refining-few-step-text-to-multiview-diffusion","slug":"refining-few-step-text-to-multiview-diffusion","title":"Refining Few-Step Text-to-Multiview Diffusion via Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.20107","repositories_listed":1,"syntology":null},{"url":"/paper/behavior-injection-preparing-language-models","slug":"behavior-injection-preparing-language-models","title":"Behavior Injection: Preparing Language Models for Reinforcement Learning","date":"2025-05-25","arxiv_id":"2505.18917","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/behavior-injection-preparing-language-models#ran","syntology_url":"https://syntology.ai/paper/2505.18917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18917"}},"official":{"repos":["czp16/bridge-llm-reasoning"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/serl-self-play-reinforcement-learning-for","slug":"serl-self-play-reinforcement-learning-for","title":"SeRL: Self-Play Reinforcement Learning for Large Language Models with Limited Data","date":"2025-05-25","arxiv_id":"2505.20347","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/serl-self-play-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2505.20347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20347"}},"official":{"repos":["wantbook-book/serl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/structured-reinforcement-learning-for-1","slug":"structured-reinforcement-learning-for-1","title":"Structured Reinforcement Learning for Combinatorial Decision-Making","date":"2025-05-25","arxiv_id":"2505.19053","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-meta-reinforcement-learning-with","slug":"bayesian-meta-reinforcement-learning-with","title":"Bayesian Meta-Reinforcement Learning with Laplace Variational Recurrent Networks","date":"2025-05-24","arxiv_id":"2505.18591","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-efficiency-and-exploration-in","slug":"enhancing-efficiency-and-exploration-in","title":"Enhancing Efficiency and Exploration in Reinforcement Learning for LLMs","date":"2025-05-24","arxiv_id":"2505.18573","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-latent-reasoning-via-reinforcement","slug":"hybrid-latent-reasoning-via-reinforcement","title":"Hybrid Latent Reasoning via Reinforcement Learning","date":"2025-05-24","arxiv_id":"2505.18454","repositories_listed":1,"syntology":null},{"url":"/paper/a-robust-ppo-optimized-tabular-transformer","slug":"a-robust-ppo-optimized-tabular-transformer","title":"A Robust PPO-optimized Tabular Transformer Framework for Intrusion Detection in Industrial IoT Systems","date":"2025-05-23","arxiv_id":"2505.18234","repositories_listed":1,"syntology":null},{"url":"/paper/co-reinforcement-learning-for-unified","slug":"co-reinforcement-learning-for-unified","title":"Co-Reinforcement Learning for Unified Multimodal Understanding and Generation","date":"2025-05-23","arxiv_id":"2505.17534","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-reinforcement-learning-for-unified#ran","syntology_url":"https://syntology.ai/paper/2505.17534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17534"}},"official":{"repos":["mm-vl/ulm-r1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/reinforcement-learning-for-ballbot-navigation","slug":"reinforcement-learning-for-ballbot-navigation","title":"Reinforcement Learning for Ballbot Navigation in Uneven Terrain","date":"2025-05-23","arxiv_id":"2505.18417","repositories_listed":1,"syntology":null},{"url":"/paper/wingpt-3-0-technical-report","slug":"wingpt-3-0-technical-report","title":"WiNGPT-3.0 Technical Report","date":"2025-05-23","arxiv_id":"2505.17387","repositories_listed":1,"syntology":null},{"url":"/paper/arpo-end-to-end-policy-optimization-for-gui","slug":"arpo-end-to-end-policy-optimization-for-gui","title":"ARPO:End-to-End Policy Optimization for GUI Agents with Experience Replay","date":"2025-05-22","arxiv_id":"2505.16282","repositories_listed":1,"syntology":{"n":16,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/arpo-end-to-end-policy-optimization-for-gui#ran","syntology_url":"https://syntology.ai/paper/2505.16282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16282"}},"official":{"repos":["dvlab-research/arpo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/conciserl-conciseness-guided-reinforcement","slug":"conciserl-conciseness-guided-reinforcement","title":"ConciseRL: Conciseness-Guided Reinforcement Learning for Efficient Reasoning Models","date":"2025-05-22","arxiv_id":"2505.17250","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conciserl-conciseness-guided-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2505.17250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17250"}},"official":{"repos":["razvandu/conciserl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fact-r1-towards-explainable-video","slug":"fact-r1-towards-explainable-video","title":"Fact-R1: Towards Explainable Video Misinformation Detection with Deep Reasoning","date":"2025-05-22","arxiv_id":"2505.16836","repositories_listed":1,"syntology":null},{"url":"/paper/got-r1-unleashing-reasoning-capability-of","slug":"got-r1-unleashing-reasoning-capability-of","title":"GoT-R1: Unleashing Reasoning Capability of MLLM for Visual Generation with Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.17022","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":6,"n_ran_checked":7,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"9 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/got-r1-unleashing-reasoning-capability-of#ran","syntology_url":"https://syntology.ai/paper/2505.17022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17022"}},"official":{"repos":["gogoduan/got-r1"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ktae-a-model-free-algorithm-to-key-tokens","slug":"ktae-a-model-free-algorithm-to-key-tokens","title":"KTAE: A Model-Free Algorithm to Key-Tokens Advantage Estimation in Mathematical Reasoning","date":"2025-05-22","arxiv_id":"2505.16826","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/ktae-a-model-free-algorithm-to-key-tokens#ran","syntology_url":"https://syntology.ai/paper/2505.16826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16826"}},"official":{"repos":["xiaolizh1/ktae"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/maximum-total-correlation-reinforcement","slug":"maximum-total-correlation-reinforcement","title":"Maximum Total Correlation Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.16734","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":6,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"8 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/maximum-total-correlation-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2505.16734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16734"}},"official":{"repos":["bangyou01/mtc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-temporal-difference-method-for-stochastic","slug":"a-temporal-difference-method-for-stochastic","title":"A Temporal Difference Method for Stochastic Continuous Dynamics","date":"2025-05-21","arxiv_id":"2505.15544","repositories_listed":1,"syntology":null},{"url":"/paper/avatarshield-visual-reinforcement-learning","slug":"avatarshield-visual-reinforcement-learning","title":"AvatarShield: Visual Reinforcement Learning for Human-Centric Video Forgery Detection","date":"2025-05-21","arxiv_id":"2505.15173","repositories_listed":1,"syntology":null},{"url":"/paper/convsearch-r1-enhancing-query-reformulation","slug":"convsearch-r1-enhancing-query-reformulation","title":"ConvSearch-R1: Enhancing Query Reformulation for Conversational Search with Reasoning via Reinforcement Learning","date":"2025-05-21","arxiv_id":"2505.15776","repositories_listed":1,"syntology":null},{"url":"/paper/hadamax-encoding-elevating-performance-in","slug":"hadamax-encoding-elevating-performance-in","title":"Hadamax Encoding: Elevating Performance in Model-Free Atari","date":"2025-05-21","arxiv_id":"2505.15345","repositories_listed":1,"syntology":null},{"url":"/paper/nover-incentive-training-for-language-models","slug":"nover-incentive-training-for-language-models","title":"NOVER: Incentive Training for Language Models via Verifier-Free Reinforcement Learning","date":"2025-05-21","arxiv_id":"2505.16022","repositories_listed":1,"syntology":null},{"url":"/paper/deepeyes-incentivizing-thinking-with-images","slug":"deepeyes-incentivizing-thinking-with-images","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14362","repositories_listed":1,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deepeyes-incentivizing-thinking-with-images#ran","syntology_url":"https://syntology.ai/paper/2505.14362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14362"}},"official":{"repos":["visual-agent/deepeyes"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/korgym-a-dynamic-game-platform-for-llm","slug":"korgym-a-dynamic-game-platform-for-llm","title":"KORGym: A Dynamic Game Platform for LLM Reasoning Evaluation","date":"2025-05-20","arxiv_id":"2505.14552","repositories_listed":1,"syntology":null},{"url":"/paper/prl-prompts-from-reinforcement-learning","slug":"prl-prompts-from-reinforcement-learning","title":"PRL: Prompts from Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14412","repositories_listed":1,"syntology":null},{"url":"/paper/rlvr-world-training-world-models-with","slug":"rlvr-world-training-world-models-with","title":"RLVR-World: Training World Models with Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.13934","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rlvr-world-training-world-models-with#ran","syntology_url":"https://syntology.ai/paper/2505.13934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13934"}},"official":{"repos":["thuml/RLVR-World"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-and-computationally-efficient-1","slug":"sample-and-computationally-efficient-1","title":"Sample and Computationally Efficient Continuous-Time Reinforcement Learning with General Function Approximation","date":"2025-05-20","arxiv_id":"2505.14821","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-moeas-for-solving-continuous","slug":"benchmarking-moeas-for-solving-continuous","title":"Benchmarking MOEAs for solving continuous multi-objective RL problems","date":"2025-05-19","arxiv_id":"2505.13726","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-explanations-for-continuous","slug":"counterfactual-explanations-for-continuous","title":"Counterfactual Explanations for Continuous Action Reinforcement Learning","date":"2025-05-19","arxiv_id":"2505.12701","repositories_listed":1,"syntology":null},{"url":"/paper/dual-agent-reinforcement-learning-for","slug":"dual-agent-reinforcement-learning-for","title":"Dual-Agent Reinforcement Learning for Automated Feature Generation","date":"2025-05-19","arxiv_id":"2505.12628","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dual-agent-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2505.12628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12628"}},"official":{"repos":["extess0/darl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/extrans-multilingual-deep-reasoning","slug":"extrans-multilingual-deep-reasoning","title":"ExTrans: Multilingual Deep Reasoning Translation via Exemplar-Enhanced Reinforcement Learning","date":"2025-05-19","arxiv_id":"2505.12996","repositories_listed":1,"syntology":null},{"url":"/paper/retrospex-language-agent-meets-offline","slug":"retrospex-language-agent-meets-offline","title":"Retrospex: Language Agent Meets Offline Reinforcement Learning Critic","date":"2025-05-17","arxiv_id":"2505.11807","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11409","slug":"2505-11409","title":"Visual Planning: Let's Think Only with Images","date":"2025-05-16","arxiv_id":"2505.11409","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2505-11409#ran","syntology_url":"https://syntology.ai/paper/2505.11409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11409"}},"official":{"repos":["yix8/visualplanning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reasoning-on-a-budget-miniaturizing-deepseek","slug":"reasoning-on-a-budget-miniaturizing-deepseek","title":"Reasoning on a Budget: Miniaturizing DeepSeek R1 with SFT-GRPO Alignment for Instruction-Tuned LLMs","date":"2025-05-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/beyond-aha-toward-systematic-meta-abilities","slug":"beyond-aha-toward-systematic-meta-abilities","title":"Beyond 'Aha!': Toward Systematic Meta-Abilities Alignment in Large Reasoning Models","date":"2025-05-15","arxiv_id":"2505.10554","repositories_listed":1,"syntology":null},{"url":"/paper/imaginebench-evaluating-reinforcement","slug":"imaginebench-evaluating-reinforcement","title":"ImagineBench: Evaluating Reinforcement Learning with Large Language Model Rollouts","date":"2025-05-15","arxiv_id":"2505.10010","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-the-known-decision-making-with","slug":"beyond-the-known-decision-making-with","title":"Beyond the Known: Decision Making with Counterfactual Reasoning Decision Transformer","date":"2025-05-14","arxiv_id":"2505.09114","repositories_listed":1,"syntology":null},{"url":"/paper/dancegrpo-unleashing-grpo-on-visual","slug":"dancegrpo-unleashing-grpo-on-visual","title":"DanceGRPO: Unleashing GRPO on Visual Generation","date":"2025-05-12","arxiv_id":"2505.07818","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dancegrpo-unleashing-grpo-on-visual#ran","syntology_url":"https://syntology.ai/paper/2505.07818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07818"}},"official":null}},{"url":"/paper/mle-dojo-interactive-environments-for","slug":"mle-dojo-interactive-environments-for","title":"MLE-Dojo: Interactive Environments for Empowering LLM Agents in Machine Learning Engineering","date":"2025-05-12","arxiv_id":"2505.07782","repositories_listed":1,"syntology":null},{"url":"/paper/structural-entropy-guided-agent-for-detecting","slug":"structural-entropy-guided-agent-for-detecting","title":"Structural Entropy Guided Agent for Detecting and Repairing Knowledge Deficiencies in LLMs","date":"2025-05-12","arxiv_id":"2505.07184","repositories_listed":1,"syntology":null},{"url":"/paper/a-critical-assessment-of-reinforcement","slug":"a-critical-assessment-of-reinforcement","title":"A critical assessment of reinforcement learning methods for microswimmer navigation in complex flows","date":"2025-05-08","arxiv_id":"2505.05525","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-and-robust-dbscan-with-multi-agent","slug":"adaptive-and-robust-dbscan-with-multi-agent","title":"Adaptive and Robust DBSCAN with Multi-agent Reinforcement Learning","date":"2025-05-07","arxiv_id":"2505.04339","repositories_listed":1,"syntology":null},{"url":"/paper/echoink-r1-exploring-audio-visual-reasoning","slug":"echoink-r1-exploring-audio-visual-reasoning","title":"EchoInk-R1: Exploring Audio-Visual Reasoning in Multimodal LLMs via Reinforcement Learning","date":"2025-05-07","arxiv_id":"2505.04623","repositories_listed":1,"syntology":null},{"url":"/paper/rlministyler-light-weight-rl-style-agent-for","slug":"rlministyler-light-weight-rl-style-agent-for","title":"RLMiniStyler: Light-weight RL Style Agent for Arbitrary Sequential Neural Style Generation","date":"2025-05-07","arxiv_id":"2505.04424","repositories_listed":1,"syntology":null},{"url":"/paper/unraveling-the-rainbow-can-value-based","slug":"unraveling-the-rainbow-can-value-based","title":"Unraveling the Rainbow: can value-based methods schedule?","date":"2025-05-06","arxiv_id":"2505.03323","repositories_listed":1,"syntology":null},{"url":"/paper/a-generalised-and-adaptable-reinforcement","slug":"a-generalised-and-adaptable-reinforcement","title":"A Generalised and Adaptable Reinforcement Learning Stopping Method","date":"2025-05-03","arxiv_id":"2505.01907","repositories_listed":1,"syntology":null},{"url":"/paper/directly-forecasting-belief-for-reinforcement","slug":"directly-forecasting-belief-for-reinforcement","title":"Directly Forecasting Belief for Reinforcement Learning with Delays","date":"2025-05-01","arxiv_id":"2505.00546","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-27","slug":"multi-agent-reinforcement-learning-for-27","title":"Multi-Agent Reinforcement Learning for Resources Allocation Optimization: A Survey","date":"2025-04-29","arxiv_id":"2504.21048","repositories_listed":1,"syntology":null},{"url":"/paper/rulebook-bringing-co-routines-to","slug":"rulebook-bringing-co-routines-to","title":"Rulebook: bringing co-routines to reinforcement learning environments","date":"2025-04-28","arxiv_id":"2504.19625","repositories_listed":1,"syntology":null},{"url":"/paper/hypercontroller-a-hyperparameter-controller","slug":"hypercontroller-a-hyperparameter-controller","title":"HyperController: A Hyperparameter Controller for Fast and Stable Training of Reinforcement Learning Neural Networks","date":"2025-04-27","arxiv_id":"2504.19382","repositories_listed":1,"syntology":null},{"url":"/paper/skywork-r1v2-multimodal-hybrid-reinforcement","slug":"skywork-r1v2-multimodal-hybrid-reinforcement","title":"Skywork R1V2: Multimodal Hybrid Reinforcement Learning for Reasoning","date":"2025-04-23","arxiv_id":"2504.16656","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/skywork-r1v2-multimodal-hybrid-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2504.16656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16656"}},"official":{"repos":["SkyworkAI/Skywork-R1V"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/compile-scene-graphs-with-reinforcement","slug":"compile-scene-graphs-with-reinforcement","title":"Compile Scene Graphs with Reinforcement Learning","date":"2025-04-18","arxiv_id":"2504.13617","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/compile-scene-graphs-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2504.13617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13617"}},"official":{"repos":["gpt4vision/r1-sgg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-from-human-feedback-4","slug":"reinforcement-learning-from-human-feedback-4","title":"Reinforcement Learning from Human Feedback","date":"2025-04-16","arxiv_id":"2504.12501","repositories_listed":1,"syntology":null},{"url":"/paper/a-pytorch-compatible-spike-encoding-framework","slug":"a-pytorch-compatible-spike-encoding-framework","title":"A PyTorch-Compatible Spike Encoding Framework for Energy-Efficient Neuromorphic Applications","date":"2025-04-15","arxiv_id":"2504.11026","repositories_listed":1,"syntology":null},{"url":"/paper/retool-reinforcement-learning-for-strategic","slug":"retool-reinforcement-learning-for-strategic","title":"ReTool: Reinforcement Learning for Strategic Tool Use in LLMs","date":"2025-04-15","arxiv_id":"2504.11536","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-sensor-steering-strategy-using-deep","slug":"adaptive-sensor-steering-strategy-using-deep","title":"Adaptive Sensor Steering Strategy Using Deep Reinforcement Learning for Dynamic Data Acquisition in Digital Twins","date":"2025-04-14","arxiv_id":"2504.10248","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reasoning-translation-via-reinforcement","slug":"deep-reasoning-translation-via-reinforcement","title":"Deep Reasoning Translation via Reinforcement Learning","date":"2025-04-14","arxiv_id":"2504.10187","repositories_listed":1,"syntology":null},{"url":"/paper/pay-attention-to-what-and-where-interpretable","slug":"pay-attention-to-what-and-where-interpretable","title":"Pay Attention to What and Where? Interpretable Feature Extractor in Vision-based Deep Reinforcement Learning","date":"2025-04-14","arxiv_id":"2504.10071","repositories_listed":1,"syntology":null},{"url":"/paper/tinyllava-video-r1-towards-smaller-lmms-for","slug":"tinyllava-video-r1-towards-smaller-lmms-for","title":"TinyLLaVA-Video-R1: Towards Smaller LMMs for Video Reasoning","date":"2025-04-13","arxiv_id":"2504.09641","repositories_listed":1,"syntology":null},{"url":"/paper/interq-a-dqn-framework-for-optimal","slug":"interq-a-dqn-framework-for-optimal","title":"InterQ: A DQN Framework for Optimal Intermittent Control","date":"2025-04-12","arxiv_id":"2504.09035","repositories_listed":1,"syntology":null},{"url":"/paper/perception-r1-pioneering-perception-policy","slug":"perception-r1-pioneering-perception-policy","title":"Perception-R1: Pioneering Perception Policy with Reinforcement Learning","date":"2025-04-10","arxiv_id":"2504.07954","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/perception-r1-pioneering-perception-policy#ran","syntology_url":"https://syntology.ai/paper/2504.07954","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07954"}},"official":{"repos":["linkangheng/pr1"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vlm-r1-a-stable-and-generalizable-r1-style","slug":"vlm-r1-a-stable-and-generalizable-r1-style","title":"VLM-R1: A Stable and Generalizable R1-style Large Vision-Language Model","date":"2025-04-10","arxiv_id":"2504.07615","repositories_listed":1,"syntology":null},{"url":"/paper/free-random-projection-for-in-context","slug":"free-random-projection-for-in-context","title":"Free Random Projection for In-Context Reinforcement Learning","date":"2025-04-09","arxiv_id":"2504.06983","repositories_listed":1,"syntology":null},{"url":"/paper/neural-motion-simulator-pushing-the-limit-of","slug":"neural-motion-simulator-pushing-the-limit-of","title":"Neural Motion Simulator: Pushing the Limit of World Models in Reinforcement Learning","date":"2025-04-09","arxiv_id":"2504.07095","repositories_listed":1,"syntology":null},{"url":"/paper/leanabell-prover-posttraining-scaling-in","slug":"leanabell-prover-posttraining-scaling-in","title":"Leanabell-Prover: Posttraining Scaling in Formal Reasoning","date":"2025-04-08","arxiv_id":"2504.06122","repositories_listed":1,"syntology":null},{"url":"/paper/robo-taxi-fleet-coordination-at-scale-via","slug":"robo-taxi-fleet-coordination-at-scale-via","title":"Robo-taxi Fleet Coordination at Scale via Reinforcement Learning","date":"2025-04-08","arxiv_id":"2504.06125","repositories_listed":1,"syntology":null},{"url":"/paper/concise-reasoning-via-reinforcement-learning","slug":"concise-reasoning-via-reinforcement-learning","title":"Concise Reasoning via Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.05185","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/concise-reasoning-via-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2504.05185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05185"}},"official":{"repos":["ai-wand/concise-reasoning"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/deep-reinforcement-learning-algorithms-for-1","slug":"deep-reinforcement-learning-algorithms-for-1","title":"Deep Reinforcement Learning Algorithms for Option Hedging","date":"2025-04-07","arxiv_id":"2504.05521","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-mixed-traffic-and-intersection","slug":"large-scale-mixed-traffic-and-intersection","title":"Large-Scale Mixed-Traffic and Intersection Control using Multi-agent Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.04691","repositories_listed":1,"syntology":null},{"url":"/paper/playing-non-embedded-card-based-games-with","slug":"playing-non-embedded-card-based-games-with","title":"Playing Non-Embedded Card-Based Games with Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.04783","repositories_listed":1,"syntology":null},{"url":"/paper/ai2stow-end-to-end-deep-reinforcement","slug":"ai2stow-end-to-end-deep-reinforcement","title":"AI2STOW: End-to-End Deep Reinforcement Learning to Construct Master Stowage Plans under Demand Uncertainty","date":"2025-04-06","arxiv_id":"2504.04469","repositories_listed":1,"syntology":null},{"url":"/paper/solving-sokoban-using-hierarchical-1","slug":"solving-sokoban-using-hierarchical-1","title":"Solving Sokoban using Hierarchical Reinforcement Learning with Landmarks","date":"2025-04-06","arxiv_id":"2504.04366","repositories_listed":1,"syntology":null},{"url":"/paper/distillation-and-refinement-of-reasoning-in","slug":"distillation-and-refinement-of-reasoning-in","title":"Distillation and Refinement of Reasoning in Small Language Models for Document Re-ranking","date":"2025-04-04","arxiv_id":"2504.03947","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-via-object","slug":"deep-reinforcement-learning-via-object","title":"Deep Reinforcement Learning via Object-Centric Attention","date":"2025-04-03","arxiv_id":"2504.03024","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-policy-gradient-reinforcement","slug":"hierarchical-policy-gradient-reinforcement","title":"Hierarchical Policy-Gradient Reinforcement Learning for Multi-Agent Shepherding Control of Non-Cohesive Targets","date":"2025-04-03","arxiv_id":"2504.02479","repositories_listed":1,"syntology":null},{"url":"/paper/low-rank-factorizations-are-indirect","slug":"low-rank-factorizations-are-indirect","title":"Low Rank Factorizations are Indirect Encodings for Deep Neuroevolution","date":"2025-04-03","arxiv_id":"2504.03037","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistic-pontryagin-s-maximum-principle","slug":"probabilistic-pontryagin-s-maximum-principle","title":"Probabilistic Pontryagin's Maximum Principle for Continuous-Time Model-Based Reinforcement Learning","date":"2025-04-03","arxiv_id":"2504.02543","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistically-safe-and-efficient-model","slug":"probabilistically-safe-and-efficient-model","title":"Probabilistically safe and efficient model-based Reinforcement Learning","date":"2025-04-01","arxiv_id":"2504.00626","repositories_listed":1,"syntology":null},{"url":"/paper/handling-delay-in-real-time-reinforcement","slug":"handling-delay-in-real-time-reinforcement","title":"Handling Delay in Real-Time Reinforcement Learning","date":"2025-03-30","arxiv_id":"2503.23478","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":7,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/handling-delay-in-real-time-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2503.23478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23478"}},"official":{"repos":["avecplezir/realtime-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/torl-scaling-tool-integrated-rl","slug":"torl-scaling-tool-integrated-rl","title":"ToRL: Scaling Tool-Integrated RL","date":"2025-03-30","arxiv_id":"2503.23383","repositories_listed":1,"syntology":null},{"url":"/paper/q-insight-understanding-image-quality-via","slug":"q-insight-understanding-image-quality-via","title":"Q-Insight: Understanding Image Quality via Visual Reinforcement Learning","date":"2025-03-28","arxiv_id":"2503.22679","repositories_listed":1,"syntology":null},{"url":"/paper/reward-design-for-reinforcement-learning","slug":"reward-design-for-reinforcement-learning","title":"Reward Design for Reinforcement Learning Agents","date":"2025-03-27","arxiv_id":"2503.21949","repositories_listed":1,"syntology":null},{"url":"/paper/neorl-2-near-real-world-benchmarks-for","slug":"neorl-2-near-real-world-benchmarks-for","title":"NeoRL-2: Near Real-World Benchmarks for Offline Reinforcement Learning with Extended Realistic Scenarios","date":"2025-03-25","arxiv_id":"2503.19267","repositories_listed":1,"syntology":null},{"url":"/paper/research-learning-to-reason-with-search-for","slug":"research-learning-to-reason-with-search-for","title":"ReSearch: Learning to Reason with Search for LLMs via Reinforcement Learning","date":"2025-03-25","arxiv_id":"2503.19470","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/research-learning-to-reason-with-search-for#ran","syntology_url":"https://syntology.ai/paper/2503.19470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19470"}},"official":null}},{"url":"/paper/continual-reinforcement-learning-for-hvac","slug":"continual-reinforcement-learning-for-hvac","title":"Continual Reinforcement Learning for HVAC Systems Control: Integrating Hypernetworks and Transfer Learning","date":"2025-03-24","arxiv_id":"2503.19212","repositories_listed":1,"syntology":null},{"url":"/paper/curriculum-rl-meets-monte-carlo-planning","slug":"curriculum-rl-meets-monte-carlo-planning","title":"Curriculum RL meets Monte Carlo Planning: Optimization of a Real World Container Management Problem","date":"2025-03-21","arxiv_id":"2503.17194","repositories_listed":1,"syntology":null},{"url":"/paper/fastcurl-curriculum-reinforcement-learning","slug":"fastcurl-curriculum-reinforcement-learning","title":"FastCuRL: Curriculum Reinforcement Learning with Progressive Context Extension for Efficient Training R1-like Reasoning Models","date":"2025-03-21","arxiv_id":"2503.17287","repositories_listed":1,"syntology":null}],"record_sha256":"36547ed3c5aaf64c617b13e400c579d17d72849f3a9940de2f0275a591a49ad8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}