{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/49","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":49,"pages_in_order":152,"rows_per_page":100,"rows":[4801,4900],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/48","next":"/task/reinforcement-learning-1/papers/50","papers":[{"url":null,"slug":"intellilung-advancing-safe-mechanical","title":"IntelliLung: Advancing Safe Mechanical Ventilation using Offline RL with Hybrid Actions and Clinically Aligned Rewards","date":"2025-06-17","arxiv_id":"2506.14375","repositories_listed":0,"syntology":null},{"url":null,"slug":"perl-permutation-enhanced-reinforcement","title":"PeRL: Permutation-Enhanced Reinforcement Learning for Interleaved Vision-Language Reasoning","date":"2025-06-17","arxiv_id":"2506.14907","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-with-exploration-an-entropy","title":"Reasoning with Exploration: An Entropy Perspective","date":"2025-06-17","arxiv_id":"2506.14758","repositories_listed":0,"syntology":null},{"url":null,"slug":"ring-lite-scalable-reasoning-via-c3po","title":"Ring-lite: Scalable Reasoning via C3PO-Stabilized Reinforcement Learning for LLMs","date":"2025-06-17","arxiv_id":"2506.14731","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-skill-discovery-through-skill","title":"Unsupervised Skill Discovery through Skill Regions Differentiation","date":"2025-06-17","arxiv_id":"2506.14420","repositories_listed":0,"syntology":null},{"url":null,"slug":"zeroth-order-optimization-is-secretly-single","title":"Zeroth-Order Optimization is Secretly Single-Step Policy Optimization","date":"2025-06-17","arxiv_id":"2506.14460","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-technical-study-into-small-reasoning","title":"A Technical Study into Small Reasoning Language Models","date":"2025-06-16","arxiv_id":"2506.13404","repositories_listed":0,"syntology":null},{"url":null,"slug":"acereason-nemotron-1-1-advancing-math-and","title":"AceReason-Nemotron 1.1: Advancing Math and Code Reasoning through SFT and RL Synergy","date":"2025-06-16","arxiv_id":"2506.13284","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-you-see-how-i-learn-human-observers","title":"Can you see how I learn? Human observers' inferences about Reinforcement Learning agents' learning processes","date":"2025-06-16","arxiv_id":"2506.13583","repositories_listed":0,"syntology":null},{"url":null,"slug":"ego-r1-chain-of-tool-thought-for-ultra-long","title":"Ego-R1: Chain-of-Tool-Thought for Ultra-Long Egocentric Video Reasoning","date":"2025-06-16","arxiv_id":"2506.13654","repositories_listed":0,"syntology":null},{"url":null,"slug":"reindsplit-reinforced-dynamic-split-learning","title":"ReinDSplit: Reinforced Dynamic Split Learning for Pest Recognition in Precision Agriculture","date":"2025-06-16","arxiv_id":"2506.13935","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-guided-mpc-for-autonomous-greenhouse","title":"RL-Guided MPC for Autonomous Greenhouse Control","date":"2025-06-16","arxiv_id":"2506.13278","repositories_listed":0,"syntology":null},{"url":null,"slug":"socratic-rl-a-novel-framework-for-efficient","title":"Socratic RL: A Novel Framework for Efficient Knowledge Acquisition through Iterative Reflection and Viewpoint Distillation","date":"2025-06-16","arxiv_id":"2506.13358","repositories_listed":0,"syntology":null},{"url":null,"slug":"staq-it-growing-neural-networks-for-policy","title":"StaQ it! Growing neural networks for Policy Mirror Descent","date":"2025-06-16","arxiv_id":"2506.13862","repositories_listed":0,"syntology":null},{"url":"/paper/the-courage-to-stop-overcoming-sunk-cost","slug":"the-courage-to-stop-overcoming-sunk-cost","title":"The Courage to Stop: Overcoming Sunk Cost Fallacy in Deep Reinforcement Learning","date":"2025-06-16","arxiv_id":"2506.13672","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-courage-to-stop-overcoming-sunk-cost#ran","syntology_url":"https://syntology.ai/paper/2506.13672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13672"}},"official":null}},{"url":null,"slug":"capo-reinforcing-consistent-reasoning-in","title":"CAPO: Reinforcing Consistent Reasoning in Medical Decision-Making","date":"2025-06-15","arxiv_id":"2506.12849","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-neuroevolution-o-ran-enhancing-the","title":"Federated Neuroevolution O-RAN: Enhancing the Robustness of Deep Reinforcement Learning xApps","date":"2025-06-15","arxiv_id":"2506.12812","repositories_listed":0,"syntology":null},{"url":null,"slug":"mm-r5-multimodal-reasoning-enhanced-reranker","title":"MM-R5: MultiModal Reasoning-Enhanced ReRanker via Reinforcement Learning for Document Retrieval","date":"2025-06-14","arxiv_id":"2506.12364","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-treatment-planning-for-interstitial","title":"Automated Treatment Planning for Interstitial HDR Brachytherapy for Locally Advanced Cervical Cancer using Deep Reinforcement Learning","date":"2025-06-13","arxiv_id":"2506.11957","repositories_listed":0,"syntology":null},{"url":null,"slug":"eliciting-reasoning-in-language-models-with","title":"Eliciting Reasoning in Language Models with Cognitive Tools","date":"2025-06-13","arxiv_id":"2506.12115","repositories_listed":0,"syntology":null},{"url":null,"slug":"learnalign-reasoning-data-selection-for","title":"LearnAlign: Reasoning Data Selection for Reinforcement Learning in Large Language Models Based on Improved Gradient Alignment","date":"2025-06-13","arxiv_id":"2506.11480","repositories_listed":0,"syntology":null},{"url":null,"slug":"reveal-self-evolving-code-agents-via","title":"ReVeal: Self-Evolving Code Agents via Iterative Generation-Verification","date":"2025-06-13","arxiv_id":"2506.11442","repositories_listed":0,"syntology":null},{"url":null,"slug":"magistral","title":"Magistral","date":"2025-06-12","arxiv_id":"2506.10910","repositories_listed":0,"syntology":null},{"url":null,"slug":"pag-multi-turn-reinforced-llm-self-correction","title":"PAG: Multi-Turn Reinforced LLM Self-Correction with Policy as Generative Verifier","date":"2025-06-12","arxiv_id":"2506.10406","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-10153","title":"Attention on flow control: transformer-based reinforcement learning for lift regulation in highly disturbed flows","date":"2025-06-11","arxiv_id":"2506.10153","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-the-role-of-artificial-1","title":"A Survey on the Role of Artificial Intelligence and Machine Learning in 6G-V2X Applications","date":"2025-06-11","arxiv_id":"2506.09512","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-treatment-planning-using","title":"Automatic Treatment Planning using Reinforcement Learning for High-dose-rate Prostate Brachytherapy","date":"2025-06-11","arxiv_id":"2506.09805","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-continuous-time-lqr-and","title":"Bridging Continuous-time LQR and Reinforcement Learning via Gradient Flow of the Bellman Error","date":"2025-06-11","arxiv_id":"2506.09685","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08463","title":"How to Provably Improve Return Conditioned Supervised Learning?","date":"2025-06-10","arxiv_id":"2506.08463","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08507","title":"MasHost Builds It All: Autonomous Multi-Agent System Directed by Reinforcement Learning","date":"2025-06-10","arxiv_id":"2506.08507","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08533","title":"Robust Evolutionary Multi-Objective Network Architecture Search for Reinforcement Learning (EMNAS-RL)","date":"2025-06-10","arxiv_id":"2506.08533","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepform-reasoning-large-language-model-for","title":"DeepForm: Reasoning Large Language Model for Communication System Formulation","date":"2025-06-10","arxiv_id":"2506.08551","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-by-random-reward-perturbation","title":"Exploration by Random Reward Perturbation","date":"2025-06-10","arxiv_id":"2506.08737","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-operating-strategy-for-pv-bess","title":"Optimal Operating Strategy for PV-BESS Households: Balancing Self-Consumption and Self-Sufficiency","date":"2025-06-10","arxiv_id":"2506.17268","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-based-trajectory-clustering-in-offline","title":"Policy-Based Trajectory Clustering in Offline Reinforcement Learning","date":"2025-06-10","arxiv_id":"2506.09202","repositories_listed":0,"syntology":null},{"url":"/paper/reinforcement-learning-teachers-of-test-time","slug":"reinforcement-learning-teachers-of-test-time","title":"Reinforcement Learning Teachers of Test Time Scaling","date":"2025-06-10","arxiv_id":"2506.08388","repositories_listed":0,"syntology":{"n":25,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":0,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/reinforcement-learning-teachers-of-test-time#ran","syntology_url":"https://syntology.ai/paper/2506.08388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08388"}},"official":null}},{"url":null,"slug":"tgrpo-fine-tuning-vision-language-action","title":"TGRPO :Fine-tuning Vision-Language-Action Model via Trajectory-wise Group Relative Policy Optimization","date":"2025-06-10","arxiv_id":"2506.08440","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-llms-reasoning-by-reinforcing","title":"AbstRaL: Augmenting LLMs' Reasoning by Reinforcing Abstract Thinking","date":"2025-06-09","arxiv_id":"2506.07751","repositories_listed":0,"syntology":null},{"url":null,"slug":"bingo-boosting-efficient-reasoning-of-llms","title":"Bingo: Boosting Efficient Reasoning of LLMs via Dynamic and Significance-based Reinforcement Learning","date":"2025-06-09","arxiv_id":"2506.08125","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralizing-multi-agent-reinforcement","title":"Decentralizing Multi-Agent Reinforcement Learning with Temporal Causal Information","date":"2025-06-09","arxiv_id":"2506.07829","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepvideo-r1-video-reinforcement-fine-tuning","title":"DeepVideo-R1: Video Reinforcement Fine-Tuning via Difficulty-aware Regressive GRPO","date":"2025-06-09","arxiv_id":"2506.07464","repositories_listed":0,"syntology":null},{"url":null,"slug":"lucifer-language-understanding-and-context","title":"LUCIFER: Language Understanding and Context-Infused Framework for Exploration and Behavior Refinement","date":"2025-06-09","arxiv_id":"2506.07915","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-pre-training","title":"Reinforcement Pre-Training","date":"2025-06-09","arxiv_id":"2506.08007","repositories_listed":0,"syntology":null},{"url":null,"slug":"through-the-valley-path-to-effective-long-cot","title":"Through the Valley: Path to Effective Long CoT Training for Small Language Models","date":"2025-06-09","arxiv_id":"2506.07712","repositories_listed":0,"syntology":null},{"url":null,"slug":"carol-context-aware-adaptation-for-robot","title":"CARoL: Context-aware Adaptation for Robot Learning","date":"2025-06-08","arxiv_id":"2506.07006","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-clarify-by-reinforcement-learning","title":"Learning to Clarify by Reinforcement Learning Through Reward-Weighted Fine-Tuning","date":"2025-06-08","arxiv_id":"2506.06964","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-generalization-of-data-assisted","title":"On the Generalization of Data-Assisted Control in port-Hamiltonian Systems (DAC-pH)","date":"2025-06-08","arxiv_id":"2506.07079","repositories_listed":0,"syntology":null},{"url":null,"slug":"qforce-rl-quantized-fpga-optimized","title":"QForce-RL: Quantized FPGA-Optimized Reinforcement Learning Compute Engine","date":"2025-06-08","arxiv_id":"2506.07046","repositories_listed":0,"syntology":null},{"url":null,"slug":"reliable-critics-monotonic-improvement-and","title":"Reliable Critics: Monotonic Improvement and Convergence Guarantees for Reinforcement Learning","date":"2025-06-08","arxiv_id":"2506.07134","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-aware-reinforcement-learning-for-1","title":"Safety-Aware Reinforcement Learning for Control via Risk-Sensitive Action-Value Iteration and Quantile Regression","date":"2025-06-08","arxiv_id":"2506.06954","repositories_listed":0,"syntology":null},{"url":null,"slug":"codecontests-high-quality-test-case","title":"CodeContests+: High-Quality Test Case Generation for Competitive Programming","date":"2025-06-06","arxiv_id":"2506.05817","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-wireless-networks-reinforced-in","title":"Prompting Wireless Networks: Reinforced In-Context Learning for Power Control","date":"2025-06-06","arxiv_id":"2506.06526","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-infant-sleep-optimized-driving","title":"Towards Infant Sleep-Optimized Driving: Synergizing Wearable and Vehicle Sensing in Intelligent Cruise Control","date":"2025-06-06","arxiv_id":"2506.06459","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-accuracy-dissecting-mathematical","title":"Beyond Accuracy: Dissecting Mathematical Reasoning for LLMs Under Reinforcement Learning","date":"2025-06-05","arxiv_id":"2506.04723","repositories_listed":0,"syntology":null},{"url":null,"slug":"confidence-is-all-you-need-few-shot-rl-fine","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","date":"2025-06-05","arxiv_id":"2506.06395","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-mechanism-of-reasoning-pattern","title":"On the Mechanism of Reasoning Pattern Selection in Reinforcement Learning for Language Models","date":"2025-06-05","arxiv_id":"2506.04695","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-optimal-q-learning-with-low-cost-for","title":"Regret-Optimal Q-Learning with Low Cost for Single-Agent and Federated Reinforcement Learning","date":"2025-06-05","arxiv_id":"2506.04626","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-planning-and-policy-optimization-via","title":"Safe Planning and Policy Optimization via World Model Learning","date":"2025-06-05","arxiv_id":"2506.04828","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lyapunov-drift-plus-penalty-method-tailored","title":"A Lyapunov Drift-Plus-Penalty Method Tailored for Reinforcement Learning with Queue Stability","date":"2025-06-04","arxiv_id":"2506.04291","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-multimodal-reasoning-from-optimized","title":"Advancing Multimodal Reasoning: From Optimized Cold Start to Staged Reinforcement Learning","date":"2025-06-04","arxiv_id":"2506.04207","repositories_listed":0,"syntology":null},{"url":null,"slug":"core-constraint-aware-one-step-reinforcement","title":"CORE: Constraint-Aware One-Step Reinforcement Learning for Simulation-Guided Neural Network Accelerator Design","date":"2025-06-04","arxiv_id":"2506.03474","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-at-criticality-in-large-language","title":"Learning-at-Criticality in Large Language Models for Quantum Field Theory and Beyond","date":"2025-06-04","arxiv_id":"2506.03703","repositories_listed":0,"syntology":null},{"url":null,"slug":"slac-simulation-pretrained-latent-action","title":"SLAC: Simulation-Pretrained Latent Action Space for Whole-Body Real-World RL","date":"2025-06-04","arxiv_id":"2506.04147","repositories_listed":0,"syntology":null},{"url":null,"slug":"critique-grpo-advancing-llm-reasoning-with","title":"Critique-GRPO: Advancing LLM Reasoning with Natural Language and Numerical Feedback","date":"2025-06-03","arxiv_id":"2506.03106","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-modeling-for-learning-decision-making","title":"Joint Modeling for Learning Decision-Making Dynamics in Behavioral Experiments","date":"2025-06-03","arxiv_id":"2506.02394","repositories_listed":0,"syntology":null},{"url":null,"slug":"learned-controllers-for-agile-quadrotors-in","title":"Learned Controllers for Agile Quadrotors in Pursuit-Evasion Games","date":"2025-06-03","arxiv_id":"2506.02849","repositories_listed":0,"syntology":null},{"url":null,"slug":"unleashing-the-reasoning-potential-of-pre","title":"Unleashing the Reasoning Potential of Pre-trained LLMs by Critique Fine-Tuning on One Problem","date":"2025-06-03","arxiv_id":"2506.03295","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-assimilated-model-informed-reinforcement","title":"Data-assimilated model-informed reinforcement learning","date":"2025-06-02","arxiv_id":"2506.01755","repositories_listed":0,"syntology":null},{"url":null,"slug":"kdrl-post-training-reasoning-llms-via-unified","title":"KDRL: Post-Training Reasoning LLMs via Unified Knowledge Distillation and Reinforcement Learning","date":"2025-06-02","arxiv_id":"2506.02208","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-or-reasoning-a-close-look-at-how","title":"Knowledge or Reasoning? A Close Look at How LLMs Think Across Domains","date":"2025-06-02","arxiv_id":"2506.02126","repositories_listed":0,"syntology":null},{"url":null,"slug":"srpo-enhancing-multimodal-llm-reasoning-via","title":"SRPO: Enhancing Multimodal LLM Reasoning via Reflection-Aware Reinforcement Learning","date":"2025-06-02","arxiv_id":"2506.01713","repositories_listed":0,"syntology":null},{"url":null,"slug":"trajectory-first-a-curriculum-for-discovering","title":"Trajectory First: A Curriculum for Discovering Diverse Policies","date":"2025-06-02","arxiv_id":"2506.01568","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for-ris","title":"A Reinforcement Learning Approach for RIS-aided Fair Communications","date":"2025-06-01","arxiv_id":"2506.06344","repositories_listed":0,"syntology":null},{"url":null,"slug":"drivemind-a-dual-vlm-based-reinforcement","title":"DriveMind: A Dual-VLM based Reinforcement Learning Framework for Autonomous Driving","date":"2025-06-01","arxiv_id":"2506.00819","repositories_listed":0,"syntology":null},{"url":null,"slug":"aria-training-language-agents-with-intention","title":"ARIA: Training Language Agents with Intention-Driven Reward Aggregation","date":"2025-05-31","arxiv_id":"2506.00539","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmedagent-rl-optimizing-multi-agent","title":"MMedAgent-RL: Optimizing Multi-Agent Collaboration for Multimodal Medical Reasoning","date":"2025-05-31","arxiv_id":"2506.00555","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-hanabi","title":"Reinforcement Learning for Hanabi","date":"2025-05-31","arxiv_id":"2506.00458","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-profit-and-fairness-in-risk-based","title":"Balancing Profit and Fairness in Risk-Based Pricing Markets","date":"2025-05-30","arxiv_id":"2506.00140","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-much-backtracking-is-enough-exploring-the","title":"How Much Backtracking is Enough? Exploring the Interplay of SFT and RL in Enhancing LLM Reasoning","date":"2025-05-30","arxiv_id":"2505.24273","repositories_listed":0,"syntology":null},{"url":null,"slug":"pangu-deepdiver-adaptive-search-intensity","title":"Pangu DeepDiver: Adaptive Search Intensity Scaling via Open-Web Reinforcement Learning","date":"2025-05-30","arxiv_id":"2505.24332","repositories_listed":0,"syntology":null},{"url":null,"slug":"proxy-target-bridging-the-gap-between","title":"Proxy Target: Bridging the Gap Between Discrete Spiking Neural Networks and Continuous Control","date":"2025-05-30","arxiv_id":"2505.24161","repositories_listed":0,"syntology":null},{"url":null,"slug":"reason-svg-hybrid-reward-rl-for-aha-moments","title":"Reason-SVG: Hybrid Reward RL for Aha-Moments in Vector Graphics Generation","date":"2025-05-30","arxiv_id":"2505.24499","repositories_listed":0,"syntology":null},{"url":null,"slug":"road-responsibility-oriented-reward-design","title":"ROAD: Responsibility-Oriented Reward Design for Reinforcement Learning in Autonomous Driving","date":"2025-05-30","arxiv_id":"2505.24317","repositories_listed":0,"syntology":null},{"url":null,"slug":"adg-ambient-diffusion-guided-dataset-recovery","title":"ADG: Ambient Diffusion-Guided Dataset Recovery for Corruption-Robust Offline Reinforcement Learning","date":"2025-05-29","arxiv_id":"2505.23871","repositories_listed":0,"syntology":null},{"url":null,"slug":"afterburner-reinforcement-learning","title":"Afterburner: Reinforcement Learning Facilitates Self-Improving Code Efficiency Optimization","date":"2025-05-29","arxiv_id":"2505.23387","repositories_listed":0,"syntology":null},{"url":null,"slug":"bigger-regularized-categorical-high-capacity","title":"Bigger, Regularized, Categorical: High-Capacity Value Functions are Efficient Multi-Task Learners","date":"2025-05-29","arxiv_id":"2505.23150","repositories_listed":0,"syntology":null},{"url":null,"slug":"composite-flow-matching-for-reinforcement","title":"Composite Flow Matching for Reinforcement Learning with Shifted-Dynamics Data","date":"2025-05-29","arxiv_id":"2505.23062","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-integrity-in-llms-via-reasoning","title":"Contextual Integrity in LLMs via Reasoning and Reinforcement Learning","date":"2025-05-29","arxiv_id":"2506.04245","repositories_listed":0,"syntology":null},{"url":null,"slug":"dip-r1-deep-inspection-and-perception-with-rl","title":"DIP-R1: Deep Inspection and Perception with RL Looking Through and Understanding Complex Scenes","date":"2025-05-29","arxiv_id":"2505.23179","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversity-aware-policy-optimization-for-large","title":"Diversity-Aware Policy Optimization for Large Language Model Reasoning","date":"2025-05-29","arxiv_id":"2505.23433","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-tuning-next-scale-visual-autoregressive","title":"Fine-Tuning Next-Scale Visual Autoregressive Models with Group Relative Policy Optimization","date":"2025-05-29","arxiv_id":"2505.23331","repositories_listed":0,"syntology":null},{"url":null,"slug":"fortune-formula-driven-reinforcement-learning","title":"Fortune: Formula-Driven Reinforcement Learning for Symbolic Table Reasoning in Language Models","date":"2025-05-29","arxiv_id":"2505.23667","repositories_listed":0,"syntology":null},{"url":null,"slug":"grower-in-the-loop-interactive-reinforcement","title":"Grower-in-the-Loop Interactive Reinforcement Learning for Greenhouse Climate Control","date":"2025-05-29","arxiv_id":"2505.23355","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-cross-domain-robust-reinforcement","title":"Hybrid Cross-domain Robust Reinforcement Learning","date":"2025-05-29","arxiv_id":"2505.23003","repositories_listed":0,"syntology":null},{"url":null,"slug":"let-s-reason-formally-natural-formal-hybrid","title":"Let's Reason Formally: Natural-Formal Hybrid Reasoning Enhances LLM's Math Capability","date":"2025-05-29","arxiv_id":"2505.23703","repositories_listed":0,"syntology":null},{"url":null,"slug":"llamarl-a-distributed-asynchronous","title":"LlamaRL: A Distributed Asynchronous Reinforcement Learning Framework for Efficient Large-scale LLM Trainin","date":"2025-05-29","arxiv_id":"2505.24034","repositories_listed":0,"syntology":null},{"url":null,"slug":"measure-gradients-not-activations-enhancing","title":"Measure gradients, not activations! Enhancing neuronal activity in deep reinforcement learning","date":"2025-05-29","arxiv_id":"2505.24061","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-better-verbalized","title":"Reinforcement Learning for Better Verbalized Confidence in Long-Form Generation","date":"2025-05-29","arxiv_id":"2505.23912","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-transcript-assisted-video","title":"Unsupervised Transcript-assisted Video Summarization and Highlight Detection","date":"2025-05-29","arxiv_id":"2505.23268","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-provable-approach-for-end-to-end-safe","title":"A Provable Approach for End-to-End Safe Reinforcement Learning","date":"2025-05-28","arxiv_id":"2505.21852","repositories_listed":0,"syntology":null}],"record_sha256":"65c888a3a4838fccd1e92441848cada0ee70817dc2dac2dee512e29082d7d118","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}