{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/45","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":45,"pages_in_order":135,"rows_per_page":100,"rows":[4401,4500],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/44","next":"/task/reinforcement-learning-2/papers/46","papers":[{"url":null,"slug":"reinforcement-learning-for-active-matter","title":"Reinforcement Learning for Active Matter","date":"2025-03-30","arxiv_id":"2503.23308","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-traffic-rule-compliance-using","title":"Predictive Traffic Rule Compliance using Reinforcement Learning","date":"2025-03-29","arxiv_id":"2503.22925","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl2grid-benchmarking-reinforcement-learning","title":"RL2Grid: Benchmarking Reinforcement Learning in Power Grid Operations","date":"2025-03-29","arxiv_id":"2503.23101","repositories_listed":0,"syntology":null},{"url":null,"slug":"crllk-constrained-reinforcement-learning-for","title":"CRLLK: Constrained Reinforcement Learning for Lane Keeping in Autonomous Driving","date":"2025-03-28","arxiv_id":"2503.22248","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-guided-sequence-weighting-for","title":"Entropy-guided sequence weighting for efficient exploration in RL-based LLM fine-tuning","date":"2025-03-28","arxiv_id":"2503.22456","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-efficient-and-1","title":"Reinforcement learning for efficient and robust multi-setpoint and multi-trajectory tracking in bioprocesses","date":"2025-03-28","arxiv_id":"2503.22409","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-machine-learning","title":"Reinforcement Learning for Machine Learning Model Deployment: Evaluating Multi-Armed Bandits in ML Ops Environments","date":"2025-03-28","arxiv_id":"2503.22595","repositories_listed":0,"syntology":null},{"url":null,"slug":"rldbf-enhancing-llms-via-reinforcement","title":"RLDBF: Enhancing LLMs Via Reinforcement Learning With DataBase FeedBack","date":"2025-03-28","arxiv_id":"2504.03713","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-offline-reinforcement-learning-3","title":"Model-Based Offline Reinforcement Learning with Adversarial Data Augmentation","date":"2025-03-26","arxiv_id":"2503.20285","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-discrete","title":"Offline Reinforcement Learning with Discrete Diffusion Skills","date":"2025-03-26","arxiv_id":"2503.20176","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-beyond-limits-advances-and-open","title":"Reasoning Beyond Limits: Advances and Open Problems for LLMs","date":"2025-03-26","arxiv_id":"2503.22732","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-aware-perturbation-optimization-for","title":"State-Aware Perturbation Optimization for Robust Deep Reinforcement Learning","date":"2025-03-26","arxiv_id":"2503.20613","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-world-models-for-bilevel","title":"Synthesizing world models for bilevel planning","date":"2025-03-26","arxiv_id":"2503.20124","repositories_listed":0,"syntology":null},{"url":null,"slug":"abstracting-geo-specific-terrains-to-scale-up","title":"Abstracting Geo-specific Terrains to Scale Up Reinforcement Learning","date":"2025-03-25","arxiv_id":"2503.20078","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-chain-of-thought-with-jensen-s","title":"Learning to chain-of-thought with Jensen's evidence lower bound","date":"2025-03-25","arxiv_id":"2503.19618","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-framework-to-rule-them-all-unifying-rl","title":"One Framework to Rule Them All: Unifying RL-Based and RL-Free Methods in RLHF","date":"2025-03-25","arxiv_id":"2503.19523","repositories_listed":0,"syntology":null},{"url":null,"slug":"adventurer-exploration-with-bigan-for-deep","title":"Adventurer: Exploration with BiGAN for Deep Reinforcement Learning","date":"2025-03-24","arxiv_id":"2503.18612","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-bounds-for-two-time-scale","title":"Finite-Time Bounds for Two-Time-Scale Stochastic Approximation with Arbitrary Norm Contractions and Markovian Noise","date":"2025-03-24","arxiv_id":"2503.18391","repositories_listed":0,"syntology":null},{"url":null,"slug":"option-discovery-using-llm-guided-semantic","title":"Option Discovery Using LLM-guided Semantic Hierarchical Reinforcement Learning","date":"2025-03-24","arxiv_id":"2503.19007","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-of-2","title":"Sample-Efficient Reinforcement Learning of Koopman eNMPC","date":"2025-03-24","arxiv_id":"2503.18787","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-multi-agent-reinforcement-learning","title":"Iterative Multi-Agent Reinforcement Learning: A Novel Approach Toward Real-World Multi-Echelon Inventory Optimization","date":"2025-03-23","arxiv_id":"2503.18201","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-sample-complexity-bounds-in-bilevel","title":"On The Sample Complexity Bounds In Bilevel Reinforcement Learning","date":"2025-03-22","arxiv_id":"2503.17644","repositories_listed":0,"syntology":null},{"url":null,"slug":"grammar-and-gameplay-aligned-rl-for-game","title":"Grammar and Gameplay-aligned RL for Game Description Generation with LLMs","date":"2025-03-20","arxiv_id":"2503.15783","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-automated-semantic-interpretability","title":"Towards Automated Semantic Interpretability in Reinforcement Learning via Vision-Language Models","date":"2025-03-20","arxiv_id":"2503.16724","repositories_listed":0,"syntology":null},{"url":null,"slug":"uas-visual-navigation-in-large-and-unseen","title":"UAS Visual Navigation in Large and Unseen Environments via a Meta Agent","date":"2025-03-20","arxiv_id":"2503.15781","repositories_listed":0,"syntology":null},{"url":null,"slug":"behaviour-discovery-and-attribution-for","title":"Behaviour Discovery and Attribution for Explainable Reinforcement Learning","date":"2025-03-19","arxiv_id":"2503.14973","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-review-of-reinforcement","title":"Comprehensive Review of Reinforcement Learning for Medical Ultrasound Imaging","date":"2025-03-19","arxiv_id":"2503.16543","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepmesh-auto-regressive-artist-mesh-creation","title":"DeepMesh: Auto-Regressive Artist-mesh Creation with Reinforcement Learning","date":"2025-03-19","arxiv_id":"2503.15265","repositories_listed":0,"syntology":null},{"url":null,"slug":"partially-observable-reinforcement-learning","title":"Partially Observable Reinforcement Learning with Memory Traces","date":"2025-03-19","arxiv_id":"2503.15200","repositories_listed":0,"syntology":null},{"url":null,"slug":"reachable-sets-based-trajectory-planning","title":"Reachable Sets-based Trajectory Planning Combining Reinforcement Learning and iLQR","date":"2025-03-19","arxiv_id":"2503.17398","repositories_listed":0,"syntology":null},{"url":null,"slug":"styleloco-generative-adversarial-distillation","title":"StyleLoco: Generative Adversarial Distillation for Natural Humanoid Robot Locomotion","date":"2025-03-19","arxiv_id":"2503.15082","repositories_listed":0,"syntology":null},{"url":null,"slug":"colson-controllable-learning-based-social","title":"COLSON: Controllable Learning-Based Social Navigation via Diffusion-Based Reinforcement Learning","date":"2025-03-18","arxiv_id":"2503.13934","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctsac-curriculum-based-transformer-soft-actor","title":"CTSAC: Curriculum-Based Transformer Soft Actor-Critic for Goal-Oriented Robot Exploration","date":"2025-03-18","arxiv_id":"2503.14254","repositories_listed":0,"syntology":null},{"url":null,"slug":"pauli-network-circuit-synthesis-with","title":"Pauli Network Circuit Synthesis with Reinforcement Learning","date":"2025-03-18","arxiv_id":"2503.14448","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-driven-transformer","title":"A Reinforcement Learning-Driven Transformer GAN for Molecular Generation","date":"2025-03-17","arxiv_id":"2503.12796","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifelong-reinforcement-learning-with-1","title":"Lifelong Reinforcement Learning with Similarity-Driven Weighting by Large Models","date":"2025-03-17","arxiv_id":"2503.12923","repositories_listed":0,"syntology":null},{"url":null,"slug":"focusing-robot-open-ended-reinforcement","title":"Focusing Robot Open-Ended Reinforcement Learning Through Users' Purposes","date":"2025-03-16","arxiv_id":"2503.12579","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-time-policy-switching-for-offline","title":"Evaluation-Time Policy Switching for Offline Reinforcement Learning","date":"2025-03-15","arxiv_id":"2503.12222","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-safe","title":"Hierarchical Reinforcement Learning for Safe Mapless Navigation with Congestion Estimation","date":"2025-03-15","arxiv_id":"2503.12036","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-deepseek-models-key-innovative","title":"A Review of DeepSeek Models' Key Innovative Techniques","date":"2025-03-14","arxiv_id":"2503.11486","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-similarity-distillation-ensemble","title":"Contextual Similarity Distillation: Ensemble Uncertainties with a Single Model","date":"2025-03-14","arxiv_id":"2503.11339","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-obstacle-avoidance-with-bounded","title":"Dynamic Obstacle Avoidance with Bounded Rationality Adversarial Reinforcement Learning","date":"2025-03-14","arxiv_id":"2503.11467","repositories_listed":0,"syntology":null},{"url":null,"slug":"sketch-to-skill-bootstrapping-robot-learning","title":"Sketch-to-Skill: Bootstrapping Robot Learning with Human Drawn Trajectory Sketches","date":"2025-03-14","arxiv_id":"2503.11918","repositories_listed":0,"syntology":null},{"url":null,"slug":"es-parkour-advanced-robot-parkour-with-bio","title":"ES-Parkour: Advanced Robot Parkour with Bio-inspired Event Camera and Spiking Neural Network","date":"2025-03-13","arxiv_id":"2503.09985","repositories_listed":0,"syntology":null},{"url":null,"slug":"prism-preference-refinement-via-implicit","title":"PRISM: Preference Refinement via Implicit Scene Modeling for 3D Vision-Language Preference-Based Reinforcement Learning","date":"2025-03-13","arxiv_id":"2503.10177","repositories_listed":0,"syntology":null},{"url":null,"slug":"rotated-bitboards-in-fusc-and-reinforcement","title":"Reinforcement Learning and Life Cycle Assessment for a Circular Economy -- Towards Progressive Computer Science","date":"2025-03-13","arxiv_id":"2503.10822","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-reinforcement-learning-safety-and","title":"Evaluating Reinforcement Learning Safety and Trustworthiness in Cyber-Physical Systems","date":"2025-03-12","arxiv_id":"2503.09388","repositories_listed":0,"syntology":null},{"url":null,"slug":"marinegym-a-high-performance-reinforcement","title":"MarineGym: A High-Performance Reinforcement Learning Platform for Underwater Robotics","date":"2025-03-12","arxiv_id":"2503.09203","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-is-all-you-need","title":"Reinforcement Learning is all You Need","date":"2025-03-12","arxiv_id":"2503.09512","repositories_listed":0,"syntology":null},{"url":null,"slug":"rule-guided-reinforcement-learning-policy","title":"Rule-Guided Reinforcement Learning Policy Evaluation and Improvement","date":"2025-03-12","arxiv_id":"2503.09270","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategyproof-reinforcement-learning-from","title":"Strategyproof Reinforcement Learning from Human Feedback","date":"2025-03-12","arxiv_id":"2503.09561","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-locomotion-transformer-with","title":"Unified Locomotion Transformer with Simultaneous Sim-to-Real Transfer for Quadrupeds","date":"2025-03-12","arxiv_id":"2503.08997","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-traffic-signal-control-through","title":"Enhancing Traffic Signal Control through Model-based Reinforcement Learning and Policy Reuse","date":"2025-03-11","arxiv_id":"2503.08728","repositories_listed":0,"syntology":null},{"url":null,"slug":"langtime-a-language-guided-unified-model-for","title":"LangTime: A Language-Guided Unified Model for Time Series Forecasting with Proximal Policy Optimization","date":"2025-03-11","arxiv_id":"2503.08271","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-with-discrete","title":"Meta-Reinforcement Learning with Discrete World Models for Adaptive Load Balancing","date":"2025-03-11","arxiv_id":"2503.08872","repositories_listed":0,"syntology":null},{"url":null,"slug":"more-unlocking-scalability-in-reinforcement","title":"MoRE: Unlocking Scalability in Reinforcement Learning for Quadruped Vision-Language-Action Models","date":"2025-03-11","arxiv_id":"2503.08007","repositories_listed":0,"syntology":null},{"url":null,"slug":"authormist-evading-ai-text-detectors-with","title":"AuthorMist: Evading AI Text Detectors with Reinforcement Learning","date":"2025-03-10","arxiv_id":"2503.08716","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-conditioned-reinforcement-learning-for-1","title":"Goal Conditioned Reinforcement Learning for Photo Finishing Tuning","date":"2025-03-10","arxiv_id":"2503.07300","repositories_listed":0,"syntology":null},{"url":null,"slug":"per-dpp-sampling-framework-and-its","title":"PER-DPP Sampling Framework and Its Application in Path Planning","date":"2025-03-10","arxiv_id":"2503.07411","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-symbolic","title":"Reinforcement Learning Based Symbolic Regression for Load Modeling","date":"2025-03-10","arxiv_id":"2503.06879","repositories_listed":0,"syntology":null},{"url":null,"slug":"research-and-design-on-intelligent","title":"Research and Design on Intelligent Recognition of Unordered Targets for Robots Based on Reinforcement Learning","date":"2025-03-10","arxiv_id":"2503.07340","repositories_listed":0,"syntology":null},{"url":null,"slug":"censoring-aware-tree-based-reinforcement","title":"Censoring-Aware Tree-Based Reinforcement Learning for Estimating Dynamic Treatment Regimes with Censored Outcomes","date":"2025-03-09","arxiv_id":"2503.06690","repositories_listed":0,"syntology":null},{"url":null,"slug":"precise-insulin-delivery-for-artificial","title":"Precise Insulin Delivery for Artificial Pancreas: A Reinforcement Learning Optimized Adaptive Fuzzy Control Approach","date":"2025-03-09","arxiv_id":"2503.06701","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-shielding-for-safe","title":"Probabilistic Shielding for Safe Reinforcement Learning","date":"2025-03-09","arxiv_id":"2503.07671","repositories_listed":0,"syntology":null},{"url":null,"slug":"impoola-the-power-of-average-pooling-for","title":"Impoola: The Power of Average Pooling for Image-Based Deep Reinforcement Learning","date":"2025-03-07","arxiv_id":"2503.05546","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-robot-collaboration-through","title":"Multi-Robot Collaboration through Reinforcement Learning and Abstract Simulation","date":"2025-03-07","arxiv_id":"2503.05092","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-reinforcement-learning-enables","title":"Multi-Task Reinforcement Learning Enables Parameter Scaling","date":"2025-03-07","arxiv_id":"2503.05126","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-weighted-flow-matching-for-offline","title":"Energy-Weighted Flow Matching for Offline Reinforcement Learning","date":"2025-03-06","arxiv_id":"2503.04975","repositories_listed":0,"syntology":null},{"url":null,"slug":"hedging-with-sparse-reward-reinforcement","title":"Hedging with Sparse Reward Reinforcement Learning","date":"2025-03-06","arxiv_id":"2503.04218","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-retention-for-continual-model-based","title":"Knowledge Retention for Continual Model-Based Reinforcement Learning","date":"2025-03-06","arxiv_id":"2503.04256","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-inverse-q-learning-from","title":"Multi-Agent Inverse Q-Learning from Demonstrations","date":"2025-03-06","arxiv_id":"2503.04679","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-correct-automata-embeddings-for","title":"Provably Correct Automata Embeddings for Optimal Automata-Conditioned Reinforcement Learning","date":"2025-03-06","arxiv_id":"2503.05042","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-autonomous-reinforcement-learning-for","title":"Towards Autonomous Reinforcement Learning for Real-World Robotic Manipulation with Large Language Models","date":"2025-03-06","arxiv_id":"2503.04280","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-implicit-preference-based-policy-fine","title":"Human Implicit Preference-Based Policy Fine-tuning for Multi-Agent Reinforcement Learning in USV Swarm","date":"2025-03-05","arxiv_id":"2503.03796","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-insights-for-efficient","title":"Probabilistic Insights for Efficient Exploration Strategies in Reinforcement Learning","date":"2025-03-05","arxiv_id":"2503.03565","repositories_listed":0,"syntology":null},{"url":null,"slug":"closing-the-intent-to-reality-gap-via","title":"Closing the Intent-to-Behavior Gap via Fulfillment Priority Logic","date":"2025-03-04","arxiv_id":"2503.05818","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-threat","title":"Reinforcement Learning-based Threat Assessment","date":"2025-03-04","arxiv_id":"2503.02612","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-01069","title":"Multi-Agent Reinforcement Learning with Long-Term Performance Objectives for Service Workforce Optimization","date":"2025-03-03","arxiv_id":"2503.01069","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-01734","title":"Adversarial Agents: Black-Box Evasion Attacks with Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.01734","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-alignments-of-lens-systems-with","title":"Active Alignments of Lens Systems with Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.02075","repositories_listed":0,"syntology":null},{"url":null,"slug":"ce-u-cross-entropy-unlearning","title":"CE-U: Cross Entropy Unlearning","date":"2025-03-03","arxiv_id":"2503.01224","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-information-enhanced-model","title":"Differentiable Information Enhanced Model-Based Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.01178","repositories_listed":0,"syntology":null},{"url":null,"slug":"dpr-diffusion-preference-based-reward-for","title":"DPR: Diffusion Preference-based Reward for Offline Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.01143","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-plasticity-in-non-stationary","title":"Improving Plasticity in Non-stationary Reinforcement Learning with Evidential Proximal Policy Optimization","date":"2025-03-03","arxiv_id":"2503.01468","repositories_listed":0,"syntology":null},{"url":null,"slug":"stone-soup-multi-target-tracking-feature","title":"Stone Soup Multi-Target Tracking Feature Extraction For Autonomous Search And Track In Deep Reinforcement Learning Environment","date":"2025-03-03","arxiv_id":"2503.01293","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-m-3-text-hf-multi-agent-reinforcement","title":"M3HF: Multi-agent Reinforcement Learning from Multi-phase Human Feedback of Mixed Quality","date":"2025-03-03","arxiv_id":"2503.02077","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-emergence-of-grammar-through","title":"The Emergence of Grammar through Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.01635","repositories_listed":0,"syntology":null},{"url":null,"slug":"ladder-self-improving-llms-through-recursive","title":"LADDER: Self-Improving LLMs Through Recursive Problem Decomposition","date":"2025-03-02","arxiv_id":"2503.00735","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-optimal-reinforcement-learning-with","title":"Minimax Optimal Reinforcement Learning with Quasi-Optimism","date":"2025-03-02","arxiv_id":"2503.00810","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-entanglement-routing-with-deep-q","title":"Adaptive Entanglement Routing with Deep Q-Networks in Quantum Networks","date":"2025-03-01","arxiv_id":"2503.02895","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-reinforcement-learning-for-virtual","title":"Scalable Reinforcement Learning for Virtual Machine Scheduling","date":"2025-03-01","arxiv_id":"2503.00537","repositories_listed":0,"syntology":null},{"url":null,"slug":"shaping-laser-pulses-with-reinforcement","title":"Shaping Laser Pulses with Reinforcement Learning","date":"2025-03-01","arxiv_id":"2503.00499","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-gymnasium-a-unified-modular-benchmark","title":"Robust Gymnasium: A Unified Modular Benchmark for Robust Reinforcement Learning","date":"2025-02-27","arxiv_id":"2502.19652","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-reinforcement-learning-for-vision","title":"Sim-to-Real Reinforcement Learning for Vision-Based Dexterous Manipulation on Humanoids","date":"2025-02-27","arxiv_id":"2502.20396","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-planning-and-reinforcement-learning","title":"Combining Planning and Reinforcement Learning for Solving Relational Multiagent Domains","date":"2025-02-26","arxiv_id":"2502.19297","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalist-world-model-pre-training-for","title":"Efficient Reinforcement Learning by Guiding Generalist World Models with Non-Curated Data","date":"2025-02-26","arxiv_id":"2502.19544","repositories_listed":0,"syntology":null},{"url":null,"slug":"recurrent-auto-encoders-for-enhanced-deep","title":"Recurrent Auto-Encoders for Enhanced Deep Reinforcement Learning in Wilderness Search and Rescue Planning","date":"2025-02-26","arxiv_id":"2502.19356","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-nesterov-accelerated-distributional","title":"Adaptive Nesterov Accelerated Distributional Deep Hedging for Efficient Volatility Risk Management","date":"2025-02-25","arxiv_id":"2502.17777","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-of-deep-reinforcement-learning-1","title":"Applications of deep reinforcement learning to urban transit network design","date":"2025-02-25","arxiv_id":"2502.17758","repositories_listed":0,"syntology":null},{"url":null,"slug":"cayleypy-rl-pathfinding-and-reinforcement","title":"CayleyPy RL: Pathfinding and Reinforcement Learning on Cayley Graphs","date":"2025-02-25","arxiv_id":"2502.18663","repositories_listed":0,"syntology":null}],"record_sha256":"32efc568a65f316778a5423b2ac311284008aba5411d7fab68996b39438da377","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}