{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/50","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":50,"pages_in_order":152,"rows_per_page":100,"rows":[4901,5000],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/49","next":"/task/reinforcement-learning-1/papers/51","papers":[{"url":null,"slug":"enhancing-study-level-inference-from-clinical","title":"Enhancing Study-Level Inference from Clinical Trial Papers via RL-based Numeric Reasoning","date":"2025-05-28","arxiv_id":"2505.22928","repositories_listed":0,"syntology":null},{"url":null,"slug":"fasttd3-simple-fast-and-capable-reinforcement","title":"FastTD3: Simple, Fast, and Capable Reinforcement Learning for Humanoid Control","date":"2025-05-28","arxiv_id":"2505.22642","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-sample-convergence-bounds-for-trust","title":"Finite-Sample Convergence Bounds for Trust Region Policy Optimization in Mean-Field Games","date":"2025-05-28","arxiv_id":"2505.22781","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximizing-confidence-alone-improves","title":"Maximizing Confidence Alone Improves Reasoning","date":"2025-05-28","arxiv_id":"2505.22660","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinflow-fine-tuning-flow-matching-policy","title":"ReinFlow: Fine-tuning Flow Matching Policy with Online Reinforcement Learning","date":"2025-05-28","arxiv_id":"2505.22094","repositories_listed":0,"syntology":null},{"url":null,"slug":"sam-r1-leveraging-sam-for-reward-feedback-in","title":"SAM-R1: Leveraging SAM for Reward Feedback in Multimodal Segmentation via Reinforcement Learning","date":"2025-05-28","arxiv_id":"2505.22596","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-offline-rl-via-efficient-and","title":"Scaling Offline RL via Efficient and Expressive Shortcut Models","date":"2025-05-28","arxiv_id":"2505.22866","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-the-performance-ceiling-in-complex","title":"Breaking the Performance Ceiling in Complex Reinforcement Learning requires Inference Strategies","date":"2025-05-27","arxiv_id":"2505.21236","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-ot-gym-a-reinforcement-learning","title":"Interactive OT Gym: A Reinforcement Learning-Based Interactive Optical tweezer (OT)-Driven Microrobotics Simulation Platform","date":"2025-05-27","arxiv_id":"2505.20751","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-optimal-treatment-strategies-for-1","title":"Learning optimal treatment strategies for intraoperative hypotension using deep reinforcement learning","date":"2025-05-27","arxiv_id":"2505.21596","repositories_listed":0,"syntology":null},{"url":null,"slug":"rendering-aware-reinforcement-learning-for","title":"Rendering-Aware Reinforcement Learning for Vector Graphics Generation","date":"2025-05-27","arxiv_id":"2505.20793","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-rlaif-curriculum-alignment-with","title":"Curriculum-RLAIF: Curriculum Alignment with Reinforcement Learning from AI Feedback","date":"2025-05-26","arxiv_id":"2505.20075","repositories_listed":0,"syntology":null},{"url":null,"slug":"done-is-better-than-perfect-unlocking","title":"Done Is Better than Perfect: Unlocking Efficient Reasoning by Structured Multi-Turn Decomposition","date":"2025-05-26","arxiv_id":"2505.19788","repositories_listed":0,"syntology":null},{"url":null,"slug":"fox-in-the-henhouse-supply-chain-backdoor","title":"Fox in the Henhouse: Supply-Chain Backdoor Attacks Against Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.19532","repositories_listed":0,"syntology":null},{"url":null,"slug":"interleaved-reasoning-for-large-language","title":"Interleaved Reasoning for Large Language Models via Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.19640","repositories_listed":0,"syntology":null},{"url":null,"slug":"meddreamer-model-based-reinforcement-learning","title":"MedDreamer: Model-Based Reinforcement Learning with Latent Imagination on Complex EHRs for Clinical Decision Support","date":"2025-05-26","arxiv_id":"2505.19785","repositories_listed":0,"syntology":null},{"url":null,"slug":"mt-3-scaling-mllm-based-text-image-machine","title":"MT$^{3}$: Scaling MLLM-based Text Image Machine Translation via Multi-Task Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.19714","repositories_listed":0,"syntology":null},{"url":null,"slug":"surrogate-assisted-evolutionary-reinforcement","title":"Surrogate-Assisted Evolutionary Reinforcement Learning Based on Autoencoder and Hyperbolic Neural Network","date":"2025-05-26","arxiv_id":"2505.19423","repositories_listed":0,"syntology":null},{"url":null,"slug":"tevir-text-to-video-reward-with-diffusion","title":"TeViR: Text-to-Video Reward with Diffusion Models for Efficient Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.19769","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlmlight-traffic-signal-control-via-vision","title":"VLMLight: Traffic Signal Control via Vision-Language Meta-Control and Dual-Branch Reasoning","date":"2025-05-26","arxiv_id":"2505.19486","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-can-rl-bring-to-vla-generalization-an","title":"What Can RL Bring to VLA Generalization? An Empirical Study","date":"2025-05-26","arxiv_id":"2505.19789","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedora-resource-allocation-for-federated","title":"FedORA: Resource Allocation for Federated Learning in ORAN using Radio Intelligent Controllers","date":"2025-05-25","arxiv_id":"2505.19211","repositories_listed":0,"syntology":null},{"url":null,"slug":"reduce-computational-cost-in-deep","title":"Reduce Computational Cost In Deep Reinforcement Learning Via Randomized Policy Learning","date":"2025-05-25","arxiv_id":"2505.19054","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-latent-reasoning-for-llm-based","title":"Reinforced Latent Reasoning for LLM-based Recommendation","date":"2025-05-25","arxiv_id":"2505.19092","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-pessimistic-reinforcement-learning","title":"Semi-pessimistic Reinforcement Learning","date":"2025-05-25","arxiv_id":"2505.19002","repositories_listed":0,"syntology":null},{"url":null,"slug":"textdiffuser-rl-efficient-and-robust-text","title":"TextDiffuser-RL: Efficient and Robust Text Layout Optimization for High-Fidelity Text-to-Image Synthesis","date":"2025-05-25","arxiv_id":"2505.19291","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-overthinker-s-diet-cutting-token-calories","title":"The Overthinker's DIET: Cutting Token Calories with DIfficulty-AwarE Training","date":"2025-05-25","arxiv_id":"2505.19217","repositories_listed":0,"syntology":null},{"url":null,"slug":"g1-teaching-llms-to-reason-on-graphs-with","title":"G1: Teaching LLMs to Reason on Graphs with Reinforcement Learning","date":"2025-05-24","arxiv_id":"2505.18499","repositories_listed":0,"syntology":null},{"url":null,"slug":"genpo-generative-diffusion-models-meet-on","title":"GenPO: Generative Diffusion Models Meet On-Policy Reinforcement Learning","date":"2025-05-24","arxiv_id":"2505.18763","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-by-guardrails-control-barrier","title":"Guided by Guardrails: Control Barrier Functions as Safety Instructors for Robotic Learning","date":"2025-05-24","arxiv_id":"2505.18858","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effect-of-negative-gradient-in-group","title":"On the Effect of Negative Gradient in Group Relative Deep Reinforcement Optimization","date":"2025-05-24","arxiv_id":"2505.18830","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-policy-but-many-worlds-a-scalable-unified","title":"One Policy but Many Worlds: A Scalable Unified Policy for Versatile Humanoid Locomotion","date":"2025-05-24","arxiv_id":"2505.18780","repositories_listed":0,"syntology":null},{"url":null,"slug":"steering-llm-reasoning-through-bias-only","title":"Steering LLM Reasoning Through Bias-Only Adaptation","date":"2025-05-24","arxiv_id":"2505.18706","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignment-and-safety-of-diffusion-models-via","title":"Alignment and Safety of Diffusion Models via Reinforcement Learning and Reward Modeling: A Survey","date":"2025-05-23","arxiv_id":"2505.17352","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-self-weighted-guidance-for-offline","title":"Diffusion Self-Weighted Guidance for Offline Reinforcement Learning","date":"2025-05-23","arxiv_id":"2505.18345","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-rl-to-see-them-all-visual-triple-unified","title":"One RL to See Them All: Visual Triple Unified Reinforcement Learning","date":"2025-05-23","arxiv_id":"2505.18129","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-speculative-decoding-for-fast","title":"Reinforcement Speculative Decoding for Fast Ranking","date":"2025-05-23","arxiv_id":"2505.20316","repositories_listed":0,"syntology":null},{"url":null,"slug":"acereason-nemotron-advancing-math-and-code","title":"AceReason-Nemotron: Advancing Math and Code Reasoning through Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.16400","repositories_listed":0,"syntology":null},{"url":null,"slug":"backdoors-in-drl-four-environments-focusing","title":"Backdoors in DRL: Four Environments Focusing on In-distribution Triggers","date":"2025-05-22","arxiv_id":"2505.17248","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-of-renewable-energy-communities-using","title":"Control of Renewable Energy Communities using AI and Real-World Data","date":"2025-05-22","arxiv_id":"2505.17321","repositories_listed":0,"syntology":null},{"url":null,"slug":"deeprec-towards-a-deep-dive-into-the-item","title":"DeepRec: Towards a Deep Dive Into the Item Space with Large Language Model Based Recommendation","date":"2025-05-22","arxiv_id":"2505.16810","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-the-implicit-multi-branch","title":"Distilling the Implicit Multi-Branch Structure in LLMs' Reasoning via Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.16142","repositories_listed":0,"syntology":null},{"url":null,"slug":"divide-fuse-conquer-eliciting-aha-moments-in","title":"Divide-Fuse-Conquer: Eliciting \"Aha Moments\" in Multi-Scenario Games","date":"2025-05-22","arxiv_id":"2505.16401","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-sampling-that-adapts-iterative-dpo","title":"Dynamic Sampling that Adapts: Iterative DPO for Self-Aware Mathematical Reasoning","date":"2025-05-22","arxiv_id":"2505.16176","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-online-rl-fine-tuning-with-offline","title":"Efficient Online RL Fine Tuning with Offline Pre-trained Policy Only","date":"2025-05-22","arxiv_id":"2505.16856","repositories_listed":0,"syntology":null},{"url":null,"slug":"find-the-fruit-designing-a-zero-shot-sim2real","title":"Find the Fruit: Designing a Zero-Shot Sim2Real Deep RL Planner for Occlusion Aware Plant Manipulation","date":"2025-05-22","arxiv_id":"2505.16547","repositories_listed":0,"syntology":null},{"url":null,"slug":"lares-latent-reasoning-for-sequential","title":"LARES: Latent Reasoning for Sequential Recommendation","date":"2025-05-22","arxiv_id":"2505.16865","repositories_listed":0,"syntology":null},{"url":null,"slug":"mesh-rft-enhancing-mesh-generation-via-fine","title":"Mesh-RFT: Enhancing Mesh Generation via Fine-grained Reinforcement Fine-Tuning","date":"2025-05-22","arxiv_id":"2505.16761","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-with-minimum","title":"Meta-reinforcement learning with minimum attention","date":"2025-05-22","arxiv_id":"2505.16741","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-guarded-safe-reinforcement-learning","title":"Offline Guarded Safe Reinforcement Learning for Medical Treatment Optimization Strategies","date":"2025-05-22","arxiv_id":"2505.16242","repositories_listed":0,"syntology":null},{"url":null,"slug":"rap-runtime-adaptive-pruning-for-llm","title":"RAP: Runtime-Adaptive Pruning for LLM Inference","date":"2025-05-22","arxiv_id":"2505.17138","repositories_listed":0,"syntology":null},{"url":"/paper/raw2drive-reinforcement-learning-with-aligned","slug":"raw2drive-reinforcement-learning-with-aligned","title":"Raw2Drive: Reinforcement Learning with Aligned World Models for End-to-End Autonomous Driving (in CARLA v2)","date":"2025-05-22","arxiv_id":"2505.16394","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-stock-transactions","title":"Reinforcement Learning for Stock Transactions","date":"2025-05-22","arxiv_id":"2505.16099","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-aware-proto-representations-in","title":"Reward-Aware Proto-Representations in Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.16217","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategically-linked-decisions-in-long-term","title":"Strategically Linked Decisions in Long-Term Planning and Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.16833","repositories_listed":0,"syntology":null},{"url":null,"slug":"think-silently-think-fast-dynamic-latent","title":"Think Silently, Think Fast: Dynamic Latent Compression of LLM Reasoning Chains","date":"2025-05-22","arxiv_id":"2505.16552","repositories_listed":0,"syntology":null},{"url":null,"slug":"vl-safe-vision-language-guided-safety-aware","title":"VL-SAFE: Vision-Language Guided Safety-Aware Reinforcement Learning with World Models for Autonomous Driving","date":"2025-05-22","arxiv_id":"2505.16377","repositories_listed":0,"syntology":null},{"url":null,"slug":"average-reward-reinforcement-learning-for-1","title":"Average Reward Reinforcement Learning for Omega-Regular and Mean-Payoff Objectives","date":"2025-05-21","arxiv_id":"2505.15693","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-focus-adaptive-visual-search-and","title":"Chain-of-Focus: Adaptive Visual Search and Zooming for Multimodal Reasoning via RL","date":"2025-05-21","arxiv_id":"2505.15436","repositories_listed":0,"syntology":null},{"url":null,"slug":"grit-teaching-mllms-to-think-with-images","title":"GRIT: Teaching MLLMs to Think with Images","date":"2025-05-21","arxiv_id":"2505.15879","repositories_listed":0,"syntology":null},{"url":null,"slug":"hcrmp-a-llm-hinted-contextual-reinforcement","title":"HCRMP: A LLM-Hinted Contextual Reinforcement Learning Framework for Autonomous Driving","date":"2025-05-21","arxiv_id":"2505.15793","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-autonomous-oversteer-control","title":"Learning-based Autonomous Oversteer Control and Collision Avoidance","date":"2025-05-21","arxiv_id":"2505.15275","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-explorer-a-plug-in-reinforcement-learning","title":"LLM-Explorer: A Plug-in Reinforcement Learning Policy Exploration Enhancement Driven by Large Language Models","date":"2025-05-21","arxiv_id":"2505.15293","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-weaks-win-single-strong-large","title":"Multiple Weaks Win Single Strong: Large Language Models Ensemble Weak Reinforcement Learning Agents into a Supreme One","date":"2025-05-21","arxiv_id":"2505.15306","repositories_listed":0,"syntology":null},{"url":null,"slug":"pass-k-policy-optimization-solving-harder","title":"Pass@K Policy Optimization: Solving Harder Reinforcement Learning Problems","date":"2025-05-21","arxiv_id":"2505.15201","repositories_listed":0,"syntology":null},{"url":null,"slug":"pixel-reasoner-incentivizing-pixel-space","title":"Pixel Reasoner: Incentivizing Pixel-Space Reasoning with Curiosity-Driven Reinforcement Learning","date":"2025-05-21","arxiv_id":"2505.15966","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-is-enough-llms-are-in-context","title":"Reward Is Enough: LLMs Are In-Context Reinforcement Learners","date":"2025-05-21","arxiv_id":"2506.06303","repositories_listed":0,"syntology":null},{"url":null,"slug":"star-r1-spacial-transformation-reasoning-by","title":"STAR-R1: Spacial TrAnsformation Reasoning by Reinforcing Multimodal LLMs","date":"2025-05-21","arxiv_id":"2505.15804","repositories_listed":0,"syntology":null},{"url":null,"slug":"stepsearch-igniting-llms-search-ability-via","title":"StepSearch: Igniting LLMs Search Ability via Step-Wise Proximal Policy Optimization","date":"2025-05-21","arxiv_id":"2505.15107","repositories_listed":0,"syntology":null},{"url":null,"slug":"thought-augmented-policy-optimization","title":"Thought-Augmented Policy Optimization: Bridging External Guidance and Internal Capabilities","date":"2025-05-21","arxiv_id":"2505.15692","repositories_listed":0,"syntology":null},{"url":"/paper/trajectory-bellman-residual-minimization-a","slug":"trajectory-bellman-residual-minimization-a","title":"Trajectory Bellman Residual Minimization: A Simple Value-Based Method for LLM Reasoning","date":"2025-05-21","arxiv_id":"2505.15311","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trajectory-bellman-residual-minimization-a#ran","syntology_url":"https://syntology.ai/paper/2505.15311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15311"}},"official":null}},{"url":null,"slug":"vard-efficient-and-dense-fine-tuning-for","title":"VARD: Efficient and Dense Fine-Tuning for Diffusion Models with Value-based RL","date":"2025-05-21","arxiv_id":"2505.15791","repositories_listed":0,"syntology":null},{"url":null,"slug":"verifybench-benchmarking-reference-based","title":"VerifyBench: Benchmarking Reference-based Reward Systems for Large Language Models","date":"2025-05-21","arxiv_id":"2505.15801","repositories_listed":0,"syntology":null},{"url":null,"slug":"viarl-adaptive-temporal-grounding-via-visual","title":"ViaRL: Adaptive Temporal Grounding via Visual Iterated Amplification Reinforcement Learning","date":"2025-05-21","arxiv_id":"2505.15447","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-can-large-reasoning-models-save-thinking","title":"When Can Large Reasoning Models Save Thinking? Mechanistic Analysis of Behavioral Divergence in Reasoning","date":"2025-05-21","arxiv_id":"2505.15276","repositories_listed":0,"syntology":null},{"url":null,"slug":"aapo-enhance-the-reasoning-capabilities-of","title":"AAPO: Enhance the Reasoning Capabilities of LLMs with Advantage Momentum","date":"2025-05-20","arxiv_id":"2505.14264","repositories_listed":0,"syntology":null},{"url":null,"slug":"bellman-operator-convergence-enhancements-in","title":"Bellman operator convergence enhancements in reinforcement learning algorithms","date":"2025-05-20","arxiv_id":"2505.14564","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-reasoner-incentivizing-reasoning","title":"Context Reasoner: Incentivizing Reasoning Capability for Contextualized Privacy and Safety Compliance via Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14585","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-reinforcement-learning-for-load","title":"Interpretable Reinforcement Learning for Load Balancing using Kolmogorov-Arnold Networks","date":"2025-05-20","arxiv_id":"2505.14459","repositories_listed":0,"syntology":null},{"url":null,"slug":"kippo-koopman-inspired-proximal-policy","title":"KIPPO: Koopman-Inspired Proximal Policy Optimization","date":"2025-05-20","arxiv_id":"2505.14566","repositories_listed":0,"syntology":null},{"url":null,"slug":"navbench-a-unified-robotics-benchmark-for","title":"NavBench: A Unified Robotics Benchmark for Reinforcement Learning-Based Autonomous Navigation","date":"2025-05-20","arxiv_id":"2505.14526","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-of-thoughts-navigating-llm-reasoning-with","title":"RL of Thoughts: Navigating LLM Reasoning with Inference-time Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14140","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-evolving-curriculum-for-llm-reasoning","title":"Self-Evolving Curriculum for LLM Reasoning","date":"2025-05-20","arxiv_id":"2505.14970","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-normalized-cut-problem-with","title":"Normalized Cut with Reinforcement Learning in Constrained Action Space","date":"2025-05-20","arxiv_id":"2505.13986","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-effective-reinforcement-learning-fine","title":"Toward Effective Reinforcement Learning Fine-Tuning for Medical VQA in Vision-Language Models","date":"2025-05-20","arxiv_id":"2505.13973","repositories_listed":0,"syntology":null},{"url":null,"slug":"univg-r1-reasoning-guided-universal-visual","title":"UniVG-R1: Reasoning Guided Universal Visual Grounding with Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14231","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-online-rl-with-offline-data-is-all","title":"Augmenting Online RL with Offline Data is All You Need: A Unified Hybrid RL Algorithm Design and Analysis","date":"2025-05-19","arxiv_id":"2505.13768","repositories_listed":0,"syntology":null},{"url":null,"slug":"dgro-enhancing-llm-reasoning-via-exploration","title":"DGRO: Enhancing LLM Reasoning via Exploration-Exploitation Control and Reward Variance Management","date":"2025-05-19","arxiv_id":"2505.12951","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-symbolic-heuristics-for-the","title":"Exploiting Symbolic Heuristics for the Synthesis of Domain-Specific Temporal Planning Guidance using Reinforcement Learning","date":"2025-05-19","arxiv_id":"2505.13372","repositories_listed":0,"syntology":null},{"url":null,"slug":"j4r-learning-to-judge-with-equivalent-initial","title":"J4R: Learning to Judge with Equivalent Initial State Group Relative Policy Optimization","date":"2025-05-19","arxiv_id":"2505.13346","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-policy-optimization-with-group-equivalent","title":"On-Policy Optimization with Group Equivalent Preference for Multi-Programming Language Understanding","date":"2025-05-19","arxiv_id":"2505.12723","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-driven-world-model-adaptation-for","title":"Policy-Driven World Model Adaptation for Robust Offline Model-based Reinforcement Learning","date":"2025-05-19","arxiv_id":"2505.13709","repositories_listed":0,"syntology":null},{"url":null,"slug":"power-allocation-for-delay-optimization-in","title":"Power Allocation for Delay Optimization in Device-to-Device Networks: A Graph Reinforcement Learning Approach","date":"2025-05-19","arxiv_id":"2505.12902","repositories_listed":0,"syntology":null},{"url":null,"slug":"step-wise-adaptive-integration-of-supervised","title":"Step-wise Adaptive Integration of Supervised Fine-tuning and Reinforcement Learning for Task-Specific LLMs","date":"2025-05-19","arxiv_id":"2505.13026","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-distance-aware-transition","title":"Temporal Distance-aware Transition Augmentation for Offline Model-based Reinforcement Learning","date":"2025-05-19","arxiv_id":"2505.13144","repositories_listed":0,"syntology":null},{"url":null,"slug":"totrl-unlock-llm-tree-of-thoughts-reasoning","title":"ToTRL: Unlock LLM Tree-of-Thoughts Reasoning Potential through Puzzles Solving","date":"2025-05-19","arxiv_id":"2505.12717","repositories_listed":0,"syntology":null},{"url":null,"slug":"your-offline-policy-is-not-trustworthy","title":"Your Offline Policy is Not Trustworthy: Bilevel Reinforcement Learning for Sequential Portfolio Optimization","date":"2025-05-19","arxiv_id":"2505.12759","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-finite-sample-analysis-of-distributionally","title":"A Finite-Sample Analysis of Distributionally Robust Average-Reward Reinforcement Learning","date":"2025-05-18","arxiv_id":"2505.12462","repositories_listed":0,"syntology":null},{"url":null,"slug":"abflownet-optimizing-antibody-antigen-binding","title":"AbFlowNet: Optimizing Antibody-Antigen Binding Energy via Diffusion-GFlowNet Fusion","date":"2025-05-18","arxiv_id":"2505.12358","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-soft-actor-critic-with","title":"Distributional Soft Actor-Critic with Harmonic Gradient for Safe and Efficient Autonomous Driving in Multi-lane Scenarios","date":"2025-05-18","arxiv_id":"2505.13532","repositories_listed":0,"syntology":null}],"record_sha256":"81be6101aced63b3f409daa01eae6c764343c0d7a38f3e9368ce83fd1d0a6585","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}