{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/42","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":42,"pages_in_order":135,"rows_per_page":100,"rows":[4101,4200],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/41","next":"/task/reinforcement-learning-2/papers/43","papers":[{"url":"/paper/strategic-dialogue-management-via-deep","slug":"strategic-dialogue-management-via-deep","title":"Strategic Dialogue Management via Deep Reinforcement Learning","date":"2015-11-25","arxiv_id":"1511.08099","repositories_listed":1,"syntology":null},{"url":"/paper/conditional-computation-in-neural-networks","slug":"conditional-computation-in-neural-networks","title":"Conditional Computation in Neural Networks for faster models","date":"2015-11-19","arxiv_id":"1511.06297","repositories_listed":1,"syntology":null},{"url":"/paper/policy-distillation","slug":"policy-distillation","title":"Policy Distillation","date":"2015-11-19","arxiv_id":"1511.06295","repositories_listed":1,"syntology":null},{"url":"/paper/deep-spatial-autoencoders-for-visuomotor","slug":"deep-spatial-autoencoders-for-visuomotor","title":"Deep Spatial Autoencoders for Visuomotor Learning","date":"2015-09-21","arxiv_id":"1509.06113","repositories_listed":1,"syntology":null},{"url":"/paper/incentivizing-exploration-in-reinforcement","slug":"incentivizing-exploration-in-reinforcement","title":"Incentivizing Exploration In Reinforcement Learning With Deep Predictive Models","date":"2015-07-03","arxiv_id":"1507.00814","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-neural-turing-machines","slug":"reinforcement-learning-neural-turing-machines","title":"Reinforcement Learning Neural Turing Machines - Revised","date":"2015-05-04","arxiv_id":"1505.00521","repositories_listed":1,"syntology":null},{"url":"/paper/deterministic-policy-gradient-algorithms","slug":"deterministic-policy-gradient-algorithms","title":"Deterministic Policy Gradient Algorithms","date":"2014-06-22","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-in-neural-networks-an-overview","slug":"deep-learning-in-neural-networks-an-overview","title":"Deep Learning in Neural Networks: An Overview","date":"2014-04-30","arxiv_id":"1404.7828","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-the-cvar-via-sampling","slug":"optimizing-the-cvar-via-sampling","title":"Optimizing the CVaR via Sampling","date":"2014-04-15","arxiv_id":"1404.3862","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-planning-and-learning-for-multiagent","slug":"scalable-planning-and-learning-for-multiagent","title":"Scalable Planning and Learning for Multiagent POMDPs: Extended Version","date":"2014-04-04","arxiv_id":"1404.1140","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-and-exploration-via-randomized","slug":"generalization-and-exploration-via-randomized","title":"Generalization and Exploration via Randomized Value Functions","date":"2014-02-04","arxiv_id":"1402.0635","repositories_listed":1,"syntology":null},{"url":"/paper/using-reinforcement-learning-to-find-an","slug":"using-reinforcement-learning-to-find-an","title":"Using reinforcement learning to find an optimal set of features","date":"2013-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/artist-agent-a-reinforcement-learning","slug":"artist-agent-a-reinforcement-learning","title":"Artist Agent: A Reinforcement Learning Approach to Automatic Stroke Generation in Oriental Ink Painting","date":"2012-06-18","arxiv_id":"1206.4634","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-actor-critic","slug":"off-policy-actor-critic","title":"Off-Policy Actor-Critic","date":"2012-05-22","arxiv_id":"1205.4839","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/off-policy-actor-critic#ran","syntology_url":"https://syntology.ai/paper/1205.4839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1205.4839"}},"official":null}},{"url":"/paper/nonlinear-inverse-reinforcement-learning-with","slug":"nonlinear-inverse-reinforcement-learning-with","title":"Nonlinear Inverse Reinforcement Learning with Gaussian Processes","date":"2011-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/an-object-oriented-representation-for","slug":"an-object-oriented-representation-for","title":"An Object-Oriented Representation for Efficient Reinforcement Learning","date":"2008-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/least-squares-policy-iteration","slug":"least-squares-policy-iteration","title":"Least-Squares Policy Iteration","date":"2003-12-04","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/between-mdps-and-semi-mdps-a-framework-for","slug":"between-mdps-and-semi-mdps-a-framework-for","title":"Between MDPs and semi-MDPs: A framework for temporal abstraction in reinforcement learning","date":"1999-08-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/simple-statistical-gradient-following","slug":"simple-statistical-gradient-following","title":"Simple Statistical Gradient-Following Algorithms for Connectionist Reinforcement Learning","date":"1992-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":null,"slug":"cuda-l1-improving-cuda-optimization-via","title":"CUDA-L1: Improving CUDA Optimization via Contrastive Reinforcement Learning","date":"2025-07-18","arxiv_id":"2507.14111","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligning-humans-and-robots-via-reinforcement","title":"Aligning Humans and Robots via Reinforcement Learning from Implicit Human Feedback","date":"2025-07-17","arxiv_id":"2507.13171","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-resource-management-in-1","title":"Autonomous Resource Management in Microservice Systems via Reinforcement Learning","date":"2025-07-17","arxiv_id":"2507.12879","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-novelty-to-imitation-self-distilled","title":"From Novelty to Imitation: Self-Distilled Rewards for Offline Reinforcement Learning","date":"2025-07-17","arxiv_id":"2507.12815","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-bellman-method-unifying","title":"Spectral Bellman Method: Unifying Representation and Exploration in RL","date":"2025-07-17","arxiv_id":"2507.13181","repositories_listed":0,"syntology":null},{"url":null,"slug":"visionthink-smart-and-efficient-vision","title":"VisionThink: Smart and Efficient Vision Language Model via Reinforcement Learning","date":"2025-07-17","arxiv_id":"2507.13348","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-explainable-reinforcement-1","title":"A Survey of Explainable Reinforcement Learning: Targets, Methods and Needs","date":"2025-07-16","arxiv_id":"2507.12599","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-on-path","title":"Distributional Reinforcement Learning on Path-dependent Options","date":"2025-07-16","arxiv_id":"2507.12657","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-reinforcement-learning-sample","title":"Improving Reinforcement Learning Sample-Efficiency using Local Approximation","date":"2025-07-16","arxiv_id":"2507.12383","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-up-rl-unlocking-diverse-reasoning-in","title":"Scaling Up RL: Unlocking Diverse Reasoning in LLMs via Prolonged Training","date":"2025-07-16","arxiv_id":"2507.12507","repositories_listed":0,"syntology":null},{"url":null,"slug":"thought-purity-defense-paradigm-for-chain-of","title":"Thought Purity: Defense Paradigm For Chain-of-Thought Attack","date":"2025-07-16","arxiv_id":"2507.12314","repositories_listed":0,"syntology":null},{"url":null,"slug":"functional-emotion-modeling-in-biomimetic","title":"Functional Emotion Modeling in Biomimetic Reinforcement Learning","date":"2025-07-15","arxiv_id":"2507.11027","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-pairwise-distance-matching-for","title":"Local Pairwise Distance Matching for Backpropagation-Free Reinforcement Learning","date":"2025-07-15","arxiv_id":"2507.11367","repositories_listed":0,"syntology":null},{"url":null,"slug":"tactical-decision-for-multi-ugv-confrontation","title":"Tactical Decision for Multi-UGV Confrontation with a Vision-Language Model-Based Commander","date":"2025-07-15","arxiv_id":"2507.11079","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-reinforcement-learning-by-planning","title":"Continual Reinforcement Learning by Planning with Online World Models","date":"2025-07-12","arxiv_id":"2507.09177","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctrls-chain-of-thought-reasoning-via-latent","title":"CTRLS: Chain-of-Thought Reasoning via Latent State-Transition","date":"2025-07-10","arxiv_id":"2507.08182","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-curiosity-to-competence-how-world-models","title":"From Curiosity to Competence: How World Models Interact with the Dynamics of Exploration","date":"2025-07-10","arxiv_id":"2507.08210","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-generals-intelligence-mastering","title":"Artificial Generals Intelligence: Mastering Generals.io with Reinforcement Learning","date":"2025-07-09","arxiv_id":"2507.06825","repositories_listed":0,"syntology":null},{"url":null,"slug":"2048-reinforcement-learning-in-a-delayed","title":"2048: Reinforcement Learning in a Delayed Reward Environment","date":"2025-07-07","arxiv_id":"2507.05465","repositories_listed":0,"syntology":null},{"url":null,"slug":"epistemically-guided-forward-backward","title":"Epistemically-guided forward-backward exploration","date":"2025-07-07","arxiv_id":"2507.05477","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-globally-speak-locally-bridging-the","title":"Learn Globally, Speak Locally: Bridging the Gaps in Multilingual Reasoning","date":"2025-07-07","arxiv_id":"2507.05418","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-continual-reinforcement-learning","title":"A Survey of Continual Reinforcement Learning","date":"2025-06-27","arxiv_id":"2506.21872","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancements-and-challenges-in-continual","title":"Advancements and Challenges in Continual Reinforcement Learning: A Comprehensive Review","date":"2025-06-27","arxiv_id":"2506.21899","repositories_listed":0,"syntology":null},{"url":null,"slug":"layer-importance-for-mathematical-reasoning","title":"Layer Importance for Mathematical Reasoning is Forged in Pre-Training and Invariant after Post-Training","date":"2025-06-27","arxiv_id":"2506.22638","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-offline-and-online-reinforcement","title":"Bridging Offline and Online Reinforcement Learning for LLMs","date":"2025-06-26","arxiv_id":"2506.21495","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-ai-for-radar-resource-management","title":"Explainable AI for Radar Resource Management: Modified LIME in Deep Reinforcement Learning","date":"2025-06-26","arxiv_id":"2506.20916","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-reinforcement-learning-trading-agent","title":"Quantum Reinforcement Learning Trading Agent for Sector Rotation in the Taiwan Stock Market","date":"2025-06-26","arxiv_id":"2506.20930","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobile-r1-towards-interactive-reinforcement","title":"Mobile-R1: Towards Interactive Reinforcement Learning for VLM-Based Mobile Agent via Task-Level Rewards","date":"2025-06-25","arxiv_id":"2506.20332","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-reinforcement-learning-for-6","title":"Multi-Objective Reinforcement Learning for Cognitive Radar Resource Management","date":"2025-06-25","arxiv_id":"2506.20853","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-principled-path-to-fitted-distributional","title":"A Principled Path to Fitted Distributional Evaluation","date":"2025-06-24","arxiv_id":"2506.20048","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-memories-to-maps-mechanisms-of-in","title":"From Memories to Maps: Mechanisms of In-Context Reinforcement Learning in Transformers","date":"2025-06-24","arxiv_id":"2506.19686","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-and-value","title":"Hierarchical Reinforcement Learning and Value Optimization for Challenging Quadruped Locomotion","date":"2025-06-24","arxiv_id":"2506.20036","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-instruction-following-policies","title":"Learning Instruction-Following Policies through Open-Ended Instruction Relabeling with Large Language Models","date":"2025-06-24","arxiv_id":"2506.20061","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-goal-conditioned-reinforcement-2","title":"Offline Goal-Conditioned Reinforcement Learning with Projective Quasimetric Planning","date":"2025-06-23","arxiv_id":"2506.18847","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-residual-reinforcement-learning","title":"Accelerating Residual Reinforcement Learning with Uncertainty Estimation","date":"2025-06-21","arxiv_id":"2506.17564","repositories_listed":0,"syntology":null},{"url":null,"slug":"no-free-lunch-rethinking-internal-feedback","title":"No Free Lunch: Rethinking Internal Feedback for LLM Reasoning","date":"2025-06-20","arxiv_id":"2506.17219","repositories_listed":0,"syntology":null},{"url":null,"slug":"grpo-care-consistency-aware-reinforcement","title":"GRPO-CARE: Consistency-Aware Reinforcement Learning for Multimodal Reasoning","date":"2025-06-19","arxiv_id":"2506.16141","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-lifelong-reinforcement-learning","title":"Multi-Task Lifelong Reinforcement Learning for Wireless Sensor Networks","date":"2025-06-19","arxiv_id":"2506.16254","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-28","title":"Multi-Agent Reinforcement Learning for Autonomous Multi-Satellite Earth Observation: A Realistic Case Study","date":"2025-06-18","arxiv_id":"2506.15207","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-policy","title":"Reinforcement Learning-Based Policy Optimisation For Heterogeneous Radio Access","date":"2025-06-18","arxiv_id":"2506.15273","repositories_listed":0,"syntology":null},{"url":null,"slug":"steering-your-diffusion-policy-with-latent","title":"Steering Your Diffusion Policy with Latent Space Reinforcement Learning","date":"2025-06-18","arxiv_id":"2506.15799","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-survey-on-underwater-acoustic","title":"A Comprehensive Survey on Underwater Acoustic Target Positioning and Tracking: Progress, Challenges, and Perspectives","date":"2025-06-17","arxiv_id":"2506.14165","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-reinforcement-learning-for","title":"Adaptive Reinforcement Learning for Unobservable Random Delays","date":"2025-06-17","arxiv_id":"2506.14411","repositories_listed":0,"syntology":null},{"url":null,"slug":"hilight-a-hierarchical-reinforcement-learning","title":"HiLight: A Hierarchical Reinforcement Learning Framework with Global Adversarial Guidance for Large-Scale Traffic Signal Control","date":"2025-06-17","arxiv_id":"2506.14391","repositories_listed":0,"syntology":null},{"url":null,"slug":"perl-permutation-enhanced-reinforcement","title":"PeRL: Permutation-Enhanced Reinforcement Learning for Interleaved Vision-Language Reasoning","date":"2025-06-17","arxiv_id":"2506.14907","repositories_listed":0,"syntology":null},{"url":null,"slug":"situational-constrained-sequential-resources","title":"Situational-Constrained Sequential Resources Allocation via Reinforcement Learning","date":"2025-06-17","arxiv_id":"2506.14125","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-guidance-accelerates-reinforcement","title":"Adaptive Guidance Accelerates Reinforcement Learning of Reasoning Models","date":"2025-06-16","arxiv_id":"2506.13923","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-temporal-structure-an-overview-of","title":"Discovering Temporal Structure: An Overview of Hierarchical Reinforcement Learning","date":"2025-06-16","arxiv_id":"2506.14045","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-reinsurance-treaty-bidding-via-multi","title":"Dynamic Reinsurance Treaty Bidding via Multi-Agent Reinforcement Learning","date":"2025-06-16","arxiv_id":"2506.13113","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-medical-vie-via-reinforcement","title":"Efficient Medical VIE via Reinforcement Learning","date":"2025-06-16","arxiv_id":"2506.13363","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-algorithm-distillation-for-continuous","title":"Scaling Algorithm Distillation for Continuous Control with Mamba","date":"2025-06-16","arxiv_id":"2506.13892","repositories_listed":0,"syntology":null},{"url":null,"slug":"socratic-rl-a-novel-framework-for-efficient","title":"Socratic RL: A Novel Framework for Efficient Knowledge Acquisition through Iterative Reflection and Viewpoint Distillation","date":"2025-06-16","arxiv_id":"2506.13358","repositories_listed":0,"syntology":null},{"url":null,"slug":"flow-based-policy-for-online-reinforcement","title":"Flow-Based Policy for Online Reinforcement Learning","date":"2025-06-15","arxiv_id":"2506.12811","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-tolerance-via-reinforcement-learning-a","title":"Noise tolerance via reinforcement: Learning a reinforced quantum dynamics","date":"2025-06-14","arxiv_id":"2506.12418","repositories_listed":0,"syntology":null},{"url":null,"slug":"relative-entropy-regularized-reinforcement","title":"Relative Entropy Regularized Reinforcement Learning for Efficient Encrypted Policy Synthesis","date":"2025-06-14","arxiv_id":"2506.12358","repositories_listed":0,"syntology":null},{"url":null,"slug":"wasserstein-barycenter-consensus-for","title":"Wasserstein-Barycenter Consensus for Cooperative Multi-Agent Reinforcement Learning","date":"2025-06-14","arxiv_id":"2506.12497","repositories_listed":0,"syntology":null},{"url":null,"slug":"reveal-self-evolving-code-agents-via","title":"ReVeal: Self-Evolving Code Agents via Iterative Generation-Verification","date":"2025-06-13","arxiv_id":"2506.11442","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-preference-based-reinforcement-2","title":"Efficient Preference-Based Reinforcement Learning: Randomized Exploration Meets Experimental Design","date":"2025-06-11","arxiv_id":"2506.09508","repositories_listed":0,"syntology":null},{"url":null,"slug":"moorl-a-framework-for-integrating-offline","title":"MOORL: A Framework for Integrating Offline-Online Reinforcement Learning","date":"2025-06-11","arxiv_id":"2506.09574","repositories_listed":0,"syntology":null},{"url":null,"slug":"synergizing-reinforcement-learning-and","title":"Synergizing Reinforcement Learning and Genetic Algorithms for Neural Combinatorial Optimization","date":"2025-06-11","arxiv_id":"2506.09404","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamical-system-optimization","title":"Dynamical System Optimization","date":"2025-06-10","arxiv_id":"2506.08340","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-based-trajectory-clustering-in-offline","title":"Policy-Based Trajectory Clustering in Offline Reinforcement Learning","date":"2025-06-10","arxiv_id":"2506.09202","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-reinforcement-learning-based","title":"Predictive reinforcement learning based adaptive PID controller","date":"2025-06-10","arxiv_id":"2506.08509","repositories_listed":0,"syntology":null},{"url":"/paper/reinforcement-learning-teachers-of-test-time","slug":"reinforcement-learning-teachers-of-test-time","title":"Reinforcement Learning Teachers of Test Time Scaling","date":"2025-06-10","arxiv_id":"2506.08388","repositories_listed":0,"syntology":{"n":25,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":0,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/reinforcement-learning-teachers-of-test-time#ran","syntology_url":"https://syntology.ai/paper/2506.08388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08388"}},"official":null}},{"url":null,"slug":"semi-gradient-dice-for-offline-constrained","title":"Semi-gradient DICE for Offline Constrained Reinforcement Learning","date":"2025-06-10","arxiv_id":"2506.08644","repositories_listed":0,"syntology":null},{"url":null,"slug":"tgrpo-fine-tuning-vision-language-action","title":"TGRPO :Fine-tuning Vision-Language-Action Model via Trajectory-wise Group Relative Policy Optimization","date":"2025-06-10","arxiv_id":"2506.08440","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-deep-reinforcement-learning-1","title":"Towards Robust Deep Reinforcement Learning against Environmental State Perturbation","date":"2025-06-10","arxiv_id":"2506.08961","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-intelligent-fault-self-healing-mechanism","title":"An Intelligent Fault Self-Healing Mechanism for Cloud AI Systems via Integration of Large Language Models and Deep Reinforcement Learning","date":"2025-06-09","arxiv_id":"2506.07411","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralizing-multi-agent-reinforcement","title":"Decentralizing Multi-Agent Reinforcement Learning with Temporal Causal Information","date":"2025-06-09","arxiv_id":"2506.07829","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairdice-fairness-driven-offline-multi","title":"FairDICE: Fairness-Driven Offline Multi-Objective Reinforcement Learning","date":"2025-06-09","arxiv_id":"2506.08062","repositories_listed":0,"syntology":null},{"url":null,"slug":"proteinzero-self-improving-protein-generation","title":"ProteinZero: Self-Improving Protein Generation via Online Reinforcement Learning","date":"2025-06-09","arxiv_id":"2506.07459","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-via-implicit-imitation","title":"Reinforcement Learning via Implicit Imitation Guidance","date":"2025-06-09","arxiv_id":"2506.07505","repositories_listed":0,"syntology":null},{"url":null,"slug":"qforce-rl-quantized-fpga-optimized","title":"QForce-RL: Quantized FPGA-Optimized Reinforcement Learning Compute Engine","date":"2025-06-08","arxiv_id":"2506.07046","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-choice-model-specification-using","title":"Improving choice model specification using reinforcement learning","date":"2025-06-06","arxiv_id":"2506.06410","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-accuracy-dissecting-mathematical","title":"Beyond Accuracy: Dissecting Mathematical Reasoning for LLMs Under Reinforcement Learning","date":"2025-06-05","arxiv_id":"2506.04723","repositories_listed":0,"syntology":null},{"url":null,"slug":"confidence-is-all-you-need-few-shot-rl-fine","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","date":"2025-06-05","arxiv_id":"2506.06395","repositories_listed":0,"syntology":null},{"url":null,"slug":"customizing-speech-recognition-model-with","title":"Customizing Speech Recognition Model with Large Language Model Feedback","date":"2025-06-05","arxiv_id":"2506.11091","repositories_listed":0,"syntology":null},{"url":null,"slug":"discounting-and-drug-seeking-in-biological","title":"Discounting and Drug Seeking in Biological Hierarchical Reinforcement Learning","date":"2025-06-05","arxiv_id":"2506.04549","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lyapunov-drift-plus-penalty-method-tailored","title":"A Lyapunov Drift-Plus-Penalty Method Tailored for Reinforcement Learning with Queue Stability","date":"2025-06-04","arxiv_id":"2506.04291","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-risk-aware-reinforcement-learning-reward","title":"A Risk-Aware Reinforcement Learning Reward for Financial Trading","date":"2025-06-04","arxiv_id":"2506.04358","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-vehicle-lateral-control-using-deep","title":"Autonomous Vehicle Lateral Control Using Deep Reinforcement Learning with MPC-PID Demonstration","date":"2025-06-04","arxiv_id":"2506.04040","repositories_listed":0,"syntology":null}],"record_sha256":"90e1622e24178aabb9fd7d7b5d35db29c018e3ca5068600d73e103c3ec645742","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}