{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/46","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":46,"pages_in_order":132,"rows_per_page":100,"rows":[4501,4600],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/45","next":"/task/reinforcement-learning/papers/47","papers":[{"url":null,"slug":"focusing-robot-open-ended-reinforcement","title":"Focusing Robot Open-Ended Reinforcement Learning Through Users' Purposes","date":"2025-03-16","arxiv_id":"2503.12579","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-time-policy-switching-for-offline","title":"Evaluation-Time Policy Switching for Offline Reinforcement Learning","date":"2025-03-15","arxiv_id":"2503.12222","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-safe","title":"Hierarchical Reinforcement Learning for Safe Mapless Navigation with Congestion Estimation","date":"2025-03-15","arxiv_id":"2503.12036","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-deepseek-models-key-innovative","title":"A Review of DeepSeek Models' Key Innovative Techniques","date":"2025-03-14","arxiv_id":"2503.11486","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-similarity-distillation-ensemble","title":"Contextual Similarity Distillation: Ensemble Uncertainties with a Single Model","date":"2025-03-14","arxiv_id":"2503.11339","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-obstacle-avoidance-with-bounded","title":"Dynamic Obstacle Avoidance with Bounded Rationality Adversarial Reinforcement Learning","date":"2025-03-14","arxiv_id":"2503.11467","repositories_listed":0,"syntology":null},{"url":null,"slug":"sketch-to-skill-bootstrapping-robot-learning","title":"Sketch-to-Skill: Bootstrapping Robot Learning with Human Drawn Trajectory Sketches","date":"2025-03-14","arxiv_id":"2503.11918","repositories_listed":0,"syntology":null},{"url":null,"slug":"es-parkour-advanced-robot-parkour-with-bio","title":"ES-Parkour: Advanced Robot Parkour with Bio-inspired Event Camera and Spiking Neural Network","date":"2025-03-13","arxiv_id":"2503.09985","repositories_listed":0,"syntology":null},{"url":null,"slug":"prism-preference-refinement-via-implicit","title":"PRISM: Preference Refinement via Implicit Scene Modeling for 3D Vision-Language Preference-Based Reinforcement Learning","date":"2025-03-13","arxiv_id":"2503.10177","repositories_listed":0,"syntology":null},{"url":null,"slug":"rotated-bitboards-in-fusc-and-reinforcement","title":"Reinforcement Learning and Life Cycle Assessment for a Circular Economy -- Towards Progressive Computer Science","date":"2025-03-13","arxiv_id":"2503.10822","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-reinforcement-learning-safety-and","title":"Evaluating Reinforcement Learning Safety and Trustworthiness in Cyber-Physical Systems","date":"2025-03-12","arxiv_id":"2503.09388","repositories_listed":0,"syntology":null},{"url":null,"slug":"marinegym-a-high-performance-reinforcement","title":"MarineGym: A High-Performance Reinforcement Learning Platform for Underwater Robotics","date":"2025-03-12","arxiv_id":"2503.09203","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-is-all-you-need","title":"Reinforcement Learning is all You Need","date":"2025-03-12","arxiv_id":"2503.09512","repositories_listed":0,"syntology":null},{"url":null,"slug":"rule-guided-reinforcement-learning-policy","title":"Rule-Guided Reinforcement Learning Policy Evaluation and Improvement","date":"2025-03-12","arxiv_id":"2503.09270","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategyproof-reinforcement-learning-from","title":"Strategyproof Reinforcement Learning from Human Feedback","date":"2025-03-12","arxiv_id":"2503.09561","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-locomotion-transformer-with","title":"Unified Locomotion Transformer with Simultaneous Sim-to-Real Transfer for Quadrupeds","date":"2025-03-12","arxiv_id":"2503.08997","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-traffic-signal-control-through","title":"Enhancing Traffic Signal Control through Model-based Reinforcement Learning and Policy Reuse","date":"2025-03-11","arxiv_id":"2503.08728","repositories_listed":0,"syntology":null},{"url":null,"slug":"langtime-a-language-guided-unified-model-for","title":"LangTime: A Language-Guided Unified Model for Time Series Forecasting with Proximal Policy Optimization","date":"2025-03-11","arxiv_id":"2503.08271","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-with-discrete","title":"Meta-Reinforcement Learning with Discrete World Models for Adaptive Load Balancing","date":"2025-03-11","arxiv_id":"2503.08872","repositories_listed":0,"syntology":null},{"url":null,"slug":"more-unlocking-scalability-in-reinforcement","title":"MoRE: Unlocking Scalability in Reinforcement Learning for Quadruped Vision-Language-Action Models","date":"2025-03-11","arxiv_id":"2503.08007","repositories_listed":0,"syntology":null},{"url":null,"slug":"authormist-evading-ai-text-detectors-with","title":"AuthorMist: Evading AI Text Detectors with Reinforcement Learning","date":"2025-03-10","arxiv_id":"2503.08716","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-conditioned-reinforcement-learning-for-1","title":"Goal Conditioned Reinforcement Learning for Photo Finishing Tuning","date":"2025-03-10","arxiv_id":"2503.07300","repositories_listed":0,"syntology":null},{"url":null,"slug":"per-dpp-sampling-framework-and-its","title":"PER-DPP Sampling Framework and Its Application in Path Planning","date":"2025-03-10","arxiv_id":"2503.07411","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-symbolic","title":"Reinforcement Learning Based Symbolic Regression for Load Modeling","date":"2025-03-10","arxiv_id":"2503.06879","repositories_listed":0,"syntology":null},{"url":null,"slug":"research-and-design-on-intelligent","title":"Research and Design on Intelligent Recognition of Unordered Targets for Robots Based on Reinforcement Learning","date":"2025-03-10","arxiv_id":"2503.07340","repositories_listed":0,"syntology":null},{"url":null,"slug":"censoring-aware-tree-based-reinforcement","title":"Censoring-Aware Tree-Based Reinforcement Learning for Estimating Dynamic Treatment Regimes with Censored Outcomes","date":"2025-03-09","arxiv_id":"2503.06690","repositories_listed":0,"syntology":null},{"url":null,"slug":"precise-insulin-delivery-for-artificial","title":"Precise Insulin Delivery for Artificial Pancreas: A Reinforcement Learning Optimized Adaptive Fuzzy Control Approach","date":"2025-03-09","arxiv_id":"2503.06701","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-shielding-for-safe","title":"Probabilistic Shielding for Safe Reinforcement Learning","date":"2025-03-09","arxiv_id":"2503.07671","repositories_listed":0,"syntology":null},{"url":null,"slug":"impoola-the-power-of-average-pooling-for","title":"Impoola: The Power of Average Pooling for Image-Based Deep Reinforcement Learning","date":"2025-03-07","arxiv_id":"2503.05546","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-robot-collaboration-through","title":"Multi-Robot Collaboration through Reinforcement Learning and Abstract Simulation","date":"2025-03-07","arxiv_id":"2503.05092","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-reinforcement-learning-enables","title":"Multi-Task Reinforcement Learning Enables Parameter Scaling","date":"2025-03-07","arxiv_id":"2503.05126","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-weighted-flow-matching-for-offline","title":"Energy-Weighted Flow Matching for Offline Reinforcement Learning","date":"2025-03-06","arxiv_id":"2503.04975","repositories_listed":0,"syntology":null},{"url":null,"slug":"hedging-with-sparse-reward-reinforcement","title":"Hedging with Sparse Reward Reinforcement Learning","date":"2025-03-06","arxiv_id":"2503.04218","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-retention-for-continual-model-based","title":"Knowledge Retention for Continual Model-Based Reinforcement Learning","date":"2025-03-06","arxiv_id":"2503.04256","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-inverse-q-learning-from","title":"Multi-Agent Inverse Q-Learning from Demonstrations","date":"2025-03-06","arxiv_id":"2503.04679","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-correct-automata-embeddings-for","title":"Provably Correct Automata Embeddings for Optimal Automata-Conditioned Reinforcement Learning","date":"2025-03-06","arxiv_id":"2503.05042","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-autonomous-reinforcement-learning-for","title":"Towards Autonomous Reinforcement Learning for Real-World Robotic Manipulation with Large Language Models","date":"2025-03-06","arxiv_id":"2503.04280","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-implicit-preference-based-policy-fine","title":"Human Implicit Preference-Based Policy Fine-tuning for Multi-Agent Reinforcement Learning in USV Swarm","date":"2025-03-05","arxiv_id":"2503.03796","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-insights-for-efficient","title":"Probabilistic Insights for Efficient Exploration Strategies in Reinforcement Learning","date":"2025-03-05","arxiv_id":"2503.03565","repositories_listed":0,"syntology":null},{"url":null,"slug":"closing-the-intent-to-reality-gap-via","title":"Closing the Intent-to-Behavior Gap via Fulfillment Priority Logic","date":"2025-03-04","arxiv_id":"2503.05818","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-threat","title":"Reinforcement Learning-based Threat Assessment","date":"2025-03-04","arxiv_id":"2503.02612","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-01069","title":"Multi-Agent Reinforcement Learning with Long-Term Performance Objectives for Service Workforce Optimization","date":"2025-03-03","arxiv_id":"2503.01069","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-01734","title":"Adversarial Agents: Black-Box Evasion Attacks with Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.01734","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-alignments-of-lens-systems-with","title":"Active Alignments of Lens Systems with Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.02075","repositories_listed":0,"syntology":null},{"url":null,"slug":"ce-u-cross-entropy-unlearning","title":"CE-U: Cross Entropy Unlearning","date":"2025-03-03","arxiv_id":"2503.01224","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-information-enhanced-model","title":"Differentiable Information Enhanced Model-Based Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.01178","repositories_listed":0,"syntology":null},{"url":null,"slug":"dpr-diffusion-preference-based-reward-for","title":"DPR: Diffusion Preference-based Reward for Offline Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.01143","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-plasticity-in-non-stationary","title":"Improving Plasticity in Non-stationary Reinforcement Learning with Evidential Proximal Policy Optimization","date":"2025-03-03","arxiv_id":"2503.01468","repositories_listed":0,"syntology":null},{"url":null,"slug":"stone-soup-multi-target-tracking-feature","title":"Stone Soup Multi-Target Tracking Feature Extraction For Autonomous Search And Track In Deep Reinforcement Learning Environment","date":"2025-03-03","arxiv_id":"2503.01293","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-m-3-text-hf-multi-agent-reinforcement","title":"M3HF: Multi-agent Reinforcement Learning from Multi-phase Human Feedback of Mixed Quality","date":"2025-03-03","arxiv_id":"2503.02077","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-emergence-of-grammar-through","title":"The Emergence of Grammar through Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.01635","repositories_listed":0,"syntology":null},{"url":null,"slug":"ladder-self-improving-llms-through-recursive","title":"LADDER: Self-Improving LLMs Through Recursive Problem Decomposition","date":"2025-03-02","arxiv_id":"2503.00735","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-optimal-reinforcement-learning-with","title":"Minimax Optimal Reinforcement Learning with Quasi-Optimism","date":"2025-03-02","arxiv_id":"2503.00810","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-entanglement-routing-with-deep-q","title":"Adaptive Entanglement Routing with Deep Q-Networks in Quantum Networks","date":"2025-03-01","arxiv_id":"2503.02895","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-reinforcement-learning-for-virtual","title":"Scalable Reinforcement Learning for Virtual Machine Scheduling","date":"2025-03-01","arxiv_id":"2503.00537","repositories_listed":0,"syntology":null},{"url":null,"slug":"shaping-laser-pulses-with-reinforcement","title":"Shaping Laser Pulses with Reinforcement Learning","date":"2025-03-01","arxiv_id":"2503.00499","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-gymnasium-a-unified-modular-benchmark","title":"Robust Gymnasium: A Unified Modular Benchmark for Robust Reinforcement Learning","date":"2025-02-27","arxiv_id":"2502.19652","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-reinforcement-learning-for-vision","title":"Sim-to-Real Reinforcement Learning for Vision-Based Dexterous Manipulation on Humanoids","date":"2025-02-27","arxiv_id":"2502.20396","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-planning-and-reinforcement-learning","title":"Combining Planning and Reinforcement Learning for Solving Relational Multiagent Domains","date":"2025-02-26","arxiv_id":"2502.19297","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalist-world-model-pre-training-for","title":"Efficient Reinforcement Learning by Guiding Generalist World Models with Non-Curated Data","date":"2025-02-26","arxiv_id":"2502.19544","repositories_listed":0,"syntology":null},{"url":null,"slug":"recurrent-auto-encoders-for-enhanced-deep","title":"Recurrent Auto-Encoders for Enhanced Deep Reinforcement Learning in Wilderness Search and Rescue Planning","date":"2025-02-26","arxiv_id":"2502.19356","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-nesterov-accelerated-distributional","title":"Adaptive Nesterov Accelerated Distributional Deep Hedging for Efficient Volatility Risk Management","date":"2025-02-25","arxiv_id":"2502.17777","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-of-deep-reinforcement-learning-1","title":"Applications of deep reinforcement learning to urban transit network design","date":"2025-02-25","arxiv_id":"2502.17758","repositories_listed":0,"syntology":null},{"url":null,"slug":"cayleypy-rl-pathfinding-and-reinforcement","title":"CayleyPy RL: Pathfinding and Reinforcement Learning on Cayley Graphs","date":"2025-02-25","arxiv_id":"2502.18663","repositories_listed":0,"syntology":null},{"url":null,"slug":"mpo-an-efficient-post-processing-framework","title":"MPO: An Efficient Post-Processing Framework for Mixing Diverse Preference Alignment","date":"2025-02-25","arxiv_id":"2502.18699","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-on-strategic-mining-in-blockchain-a","title":"Survey on Strategic Mining in Blockchain: A Reinforcement Learning Approach","date":"2025-02-24","arxiv_id":"2502.17307","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-reinforcement-learning-for","title":"Towards Reinforcement Learning for Exploration of Speculative Execution Vulnerabilities","date":"2025-02-24","arxiv_id":"2502.16756","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-sentiment-manipulation-by-llm","title":"Exploring Sentiment Manipulation by LLM-Enabled Intelligent Trading Agents","date":"2025-02-22","arxiv_id":"2502.16343","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-user-level-private-reinforcement","title":"Towards User-level Private Reinforcement Learning with Human Feedback","date":"2025-02-22","arxiv_id":"2502.17515","repositories_listed":0,"syntology":null},{"url":"/paper/hyperspherical-normalization-for-scalable","slug":"hyperspherical-normalization-for-scalable","title":"Hyperspherical Normalization for Scalable Deep Reinforcement Learning","date":"2025-02-21","arxiv_id":"2502.15280","repositories_listed":0,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hyperspherical-normalization-for-scalable#ran","syntology_url":"https://syntology.ai/paper/2502.15280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15280"}},"official":null}},{"url":null,"slug":"the-evolving-landscape-of-llm-and-vlm","title":"The Evolving Landscape of LLM- and VLM-Integrated Reinforcement Learning","date":"2025-02-21","arxiv_id":"2502.15214","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-reward-free-reinforcement-learning","title":"Towards a Reward-Free Reinforcement Learning Framework for Vehicle Control","date":"2025-02-21","arxiv_id":"2502.15262","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-mean-field-multi-agent-reinforcement","title":"Causal Mean Field Multi-Agent Reinforcement Learning","date":"2025-02-20","arxiv_id":"2502.14200","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-q-learning-an-ill-posed-problem","title":"Is Q-learning an Ill-posed Problem?","date":"2025-02-20","arxiv_id":"2502.14365","repositories_listed":0,"syntology":null},{"url":null,"slug":"mrl-discovering-transient-execution","title":"μRL: Discovering Transient Execution Vulnerabilities Using Reinforcement Learning","date":"2025-02-20","arxiv_id":"2502.14307","repositories_listed":0,"syntology":null},{"url":null,"slug":"sprig-stackelberg-perception-reinforcement","title":"SPRIG: Stackelberg Perception-Reinforcement Learning with Internal Game Dynamics","date":"2025-02-20","arxiv_id":"2502.14264","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-target-radar-search-and-track-using","title":"Multi-Target Radar Search and Track Using Sequence-Capable Deep Reinforcement Learning","date":"2025-02-19","arxiv_id":"2502.13584","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-quantification-for-markov-chains","title":"Uncertainty quantification for Markov chains with application to temporal difference learning","date":"2025-02-19","arxiv_id":"2502.13822","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-learning-conversational-ai-a","title":"Continuous Learning Conversational AI: A Personalized Agent Framework via A2C Reinforcement Learning","date":"2025-02-18","arxiv_id":"2502.12876","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-repair-with-reinforcement-learning","title":"Implicit Repair with Reinforcement Learning in Emergent Communication","date":"2025-02-18","arxiv_id":"2502.12624","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-transformers-as-iterative","title":"Self-Supervised Transformers as Iterative Solution Improvers for Constraint Satisfaction","date":"2025-02-18","arxiv_id":"2502.15794","repositories_listed":0,"syntology":null},{"url":null,"slug":"theorem-prover-as-a-judge-for-synthetic-data","title":"Theorem Prover as a Judge for Synthetic Data Generation","date":"2025-02-18","arxiv_id":"2502.13137","repositories_listed":0,"syntology":null},{"url":null,"slug":"fitlight-federated-imitation-learning-for","title":"FitLight: Federated Imitation Learning for Plug-and-Play Autonomous Traffic Signal Control","date":"2025-02-17","arxiv_id":"2502.11937","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-mobile-ai-generated-content","title":"Intelligent Mobile AI-Generated Content Services via Interactive Prompt Engineering and Dynamic Service Provisioning","date":"2025-02-17","arxiv_id":"2502.11386","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-reason-at-the-frontier-of","title":"Learning to Reason at the Frontier of Learnability","date":"2025-02-17","arxiv_id":"2502.12272","repositories_listed":0,"syntology":null},{"url":null,"slug":"theoretical-barriers-in-bellman-based","title":"Theoretical Barriers in Bellman-Based Reinforcement Learning","date":"2025-02-17","arxiv_id":"2502.11968","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-online-resource-constrained","title":"Solving Online Resource-Constrained Scheduling for Follow-Up Observation in Astronomy: a Reinforcement Learning Approach","date":"2025-02-16","arxiv_id":"2502.11134","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-tutorial-on-llm-reasoning-relevant-methods","title":"A Tutorial on LLM Reasoning: Relevant Methods behind ChatGPT o1","date":"2025-02-15","arxiv_id":"2502.10867","repositories_listed":0,"syntology":null},{"url":null,"slug":"rule-bottleneck-reinforcement-learning-joint","title":"Rule-Bottleneck Reinforcement Learning: Joint Explanation and Decision Optimization for Resource Allocation with Language Agents","date":"2025-02-15","arxiv_id":"2502.10732","repositories_listed":0,"syntology":null},{"url":null,"slug":"tackling-the-zero-shot-reinforcement-learning","title":"Tackling the Zero-Shot Reinforcement Learning Loss Directly","date":"2025-02-15","arxiv_id":"2502.10792","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-information-prioritization-for","title":"Causal Information Prioritization for Efficient Reinforcement Learning","date":"2025-02-14","arxiv_id":"2502.10097","repositories_listed":0,"syntology":null},{"url":null,"slug":"combinatorial-reinforcement-learning-with","title":"Combinatorial Reinforcement Learning with Preference Feedback","date":"2025-02-14","arxiv_id":"2502.10158","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-we-need-to-verify-step-by-step-rethinking","title":"Do We Need to Verify Step by Step? Rethinking Process Supervision from a Theoretical Perspective","date":"2025-02-14","arxiv_id":"2502.10581","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-reinforcement-learning-for-actors","title":"Dynamic Reinforcement Learning for Actors","date":"2025-02-14","arxiv_id":"2502.10200","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-constrained","title":"Reinforcement Learning based Constrained Optimal Control: an Interpretable Reward Design","date":"2025-02-14","arxiv_id":"2502.10187","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-strategy-based-and","title":"Reinforcement Learning in Strategy-Based and Atari Games: A Review of Google DeepMinds Innovations","date":"2025-02-14","arxiv_id":"2502.10303","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-consistent-model-based-adaptation-for","title":"Self-Consistent Model-based Adaptation for Visual Reinforcement Learning","date":"2025-02-14","arxiv_id":"2502.09923","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-reinforcement-learning-for","title":"A Survey of Reinforcement Learning for Optimization in Automation","date":"2025-02-13","arxiv_id":"2502.09417","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-off-policy-n-step-td-learning","title":"Analysis of Off-Policy $n$-Step TD-Learning with Linear Function Approximation","date":"2025-02-13","arxiv_id":"2502.08941","repositories_listed":0,"syntology":null},{"url":null,"slug":"coupled-rendezvous-and-docking-maneuver","title":"Coupled Rendezvous and Docking Maneuver control of satellite using Reinforcement learning-based Adaptive Fixed-Time Sliding Mode Controller","date":"2025-02-13","arxiv_id":"2502.09517","repositories_listed":0,"syntology":null}],"record_sha256":"e740779b20e222ba4054234e15942b7f1b17cb9a8afe91c78dce50f09d99f1f1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}