{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/45","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":45,"pages_in_order":132,"rows_per_page":100,"rows":[4401,4500],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/44","next":"/task/reinforcement-learning/papers/46","papers":[{"url":null,"slug":"ai-recommendation-systems-for-lane-changing","title":"AI Recommendation Systems for Lane-Changing Using Adherence-Aware Reinforcement Learning","date":"2025-04-28","arxiv_id":"2504.20187","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-in-multi","title":"Hierarchical Reinforcement Learning in Multi-Goal Spatial Navigation with Autonomous Mobile Robots","date":"2025-04-26","arxiv_id":"2504.18794","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundations-of-safe-online-reinforcement","title":"Foundations of Safe Online Reinforcement Learning in the Linear Quadratic Regulator: $\\sqrt{T}$-Regret","date":"2025-04-25","arxiv_id":"2504.18657","repositories_listed":0,"syntology":null},{"url":null,"slug":"nemotron-research-tool-n1-exploring-tool","title":"Nemotron-Research-Tool-N1: Exploring Tool-Using Language Models with Reinforced Reasoning","date":"2025-04-25","arxiv_id":"2505.00024","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-performance-reinforcement-learning-on","title":"High-Performance Reinforcement Learning on Spot: Optimizing Simulation Parameters with Distributional Measures","date":"2025-04-24","arxiv_id":"2504.17857","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-large-language-models-to-reason-via","title":"Training Large Language Models to Reason via EM Policy Gradient","date":"2025-04-24","arxiv_id":"2504.18587","repositories_listed":0,"syntology":null},{"url":null,"slug":"anytime-safe-reinforcement-learning","title":"Anytime Safe Reinforcement Learning","date":"2025-04-23","arxiv_id":"2504.16417","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-framework-for-the","title":"Reinforcement learning framework for the mechanical design of microelectronic components under multiphysics constraints","date":"2025-04-23","arxiv_id":"2504.17142","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparks-of-tabular-reasoning-via-text2sql","title":"Sparks of Tabular Reasoning via Text2SQL Reinforcement Learning","date":"2025-04-23","arxiv_id":"2505.00016","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-physical-video-generation-with","title":"Reasoning Physical Video Generation with Diffusion Timestep Tokens via Reinforcement Learning","date":"2025-04-22","arxiv_id":"2504.15932","repositories_listed":0,"syntology":null},{"url":null,"slug":"sari-structured-audio-reasoning-via","title":"SARI: Structured Audio Reasoning via Curriculum-Guided Reinforcement Learning","date":"2025-04-22","arxiv_id":"2504.15900","repositories_listed":0,"syntology":null},{"url":null,"slug":"slim-gym-reinforcement-learning-for","title":"SLiM-Gym: Reinforcement Learning for Population Genetics","date":"2025-04-22","arxiv_id":"2504.16301","repositories_listed":0,"syntology":null},{"url":null,"slug":"lapp-large-language-model-feedback-for","title":"LAPP: Large Language Model Feedback for Preference-Driven Reinforcement Learning","date":"2025-04-21","arxiv_id":"2504.15472","repositories_listed":0,"syntology":null},{"url":null,"slug":"otc-optimal-tool-calls-via-reinforcement","title":"OTC: Optimal Tool Calls via Reinforcement Learning","date":"2025-04-21","arxiv_id":"2504.14870","repositories_listed":0,"syntology":null},{"url":null,"slug":"exposing-the-copycat-problem-of-imitation","title":"Exposing the Copycat Problem of Imitation-based Planner: A Novel Closed-Loop Simulator, Causal Benchmark and Joint IL-RL Baseline","date":"2025-04-20","arxiv_id":"2504.14709","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-from-multi-level-and","title":"Reinforcement Learning from Multi-level and Episodic Human Feedback","date":"2025-04-20","arxiv_id":"2504.14732","repositories_listed":0,"syntology":null},{"url":null,"slug":"surrogate-fitness-metrics-for-interpretable","title":"Surrogate Fitness Metrics for Interpretable Reinforcement Learning","date":"2025-04-20","arxiv_id":"2504.14645","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-lattice-boltzmann-closures-through","title":"Optimal Lattice Boltzmann Closures through Multi-Agent Reinforcement Learning","date":"2025-04-19","arxiv_id":"2504.14422","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-enhanced-reinforcement-learning-for-1","title":"Quantum-Enhanced Reinforcement Learning for Power Grid Security Assessment","date":"2025-04-19","arxiv_id":"2504.14412","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalizing-exposure-therapy-via","title":"Personalizing Exposure Therapy via Reinforcement Learning","date":"2025-04-18","arxiv_id":"2504.14095","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-optimal-discriminator-weighted-imitation","title":"An Optimal Discriminator Weighted Imitation Perspective for Reinforcement Learning","date":"2025-04-17","arxiv_id":"2504.13368","repositories_listed":0,"syntology":null},{"url":null,"slug":"instructrag-leveraging-retrieval-augmented","title":"InstructRAG: Leveraging Retrieval-Augmented Generation on Instruction Graphs for LLM-Based Task Planning","date":"2025-04-17","arxiv_id":"2504.13032","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-simulation","title":"Multi-Agent Reinforcement Learning Simulation for Environmental Policy Synthesis","date":"2025-04-17","arxiv_id":"2504.12777","repositories_listed":0,"syntology":null},{"url":null,"slug":"recursive-deep-inverse-reinforcement-learning","title":"Recursive Deep Inverse Reinforcement Learning","date":"2025-04-17","arxiv_id":"2504.13241","repositories_listed":0,"syntology":null},{"url":"/paper/vipo-value-function-inconsistency-penalized","slug":"vipo-value-function-inconsistency-penalized","title":"VIPO: Value Function Inconsistency Penalized Offline Reinforcement Learning","date":"2025-04-16","arxiv_id":"2504.11944","repositories_listed":0,"syntology":{"n":14,"n_ran":10,"n_constructed":9,"n_ran_checked":9,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":14,"phrase":"10 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vipo-value-function-inconsistency-penalized#ran","syntology_url":"https://syntology.ai/paper/2504.11944","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11944"}},"official":null}},{"url":null,"slug":"multi-agent-reinforcement-learning-for-25","title":"Multi-Agent Reinforcement Learning for Greenhouse Gas Offset Credit Markets","date":"2025-04-15","arxiv_id":"2504.11258","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-based-driving-agent-for-race-car","title":"Vision based driving agent for race car simulation environments","date":"2025-04-14","arxiv_id":"2504.10266","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-implementation-of-reinforcement","title":"Efficient Implementation of Reinforcement Learning over Homomorphic Encryption","date":"2025-04-12","arxiv_id":"2504.09335","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-more-efficient-robust-instance","title":"Towards More Efficient, Robust, Instance-adaptive, and Generalizable Sequential Decision making","date":"2025-04-12","arxiv_id":"2504.09192","repositories_listed":0,"syntology":null},{"url":null,"slug":"belief-states-for-cooperative-multi-agent","title":"Belief States for Cooperative Multi-Agent Reinforcement Learning under Partial Observability","date":"2025-04-11","arxiv_id":"2504.08417","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-driven-autonomous-rover-navigation-in","title":"Near-Driven Autonomous Rover Navigation in Complex Environments: Extensions to Urban Search-and-Rescue and Industrial Inspection","date":"2025-04-11","arxiv_id":"2504.17794","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-driven-plant-wide","title":"Reinforcement Learning-Driven Plant-Wide Refinery Planning Using Model Decomposition","date":"2025-04-11","arxiv_id":"2504.08642","repositories_listed":0,"syntology":null},{"url":null,"slug":"relaxing-the-markov-requirements-on","title":"A Framework of decision-relevant observability: Reinforcement Learning converges under relative ignorability","date":"2025-04-10","arxiv_id":"2504.07722","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-the-foundations-for-continual","title":"Rethinking the Foundations for Continual Reinforcement Learning","date":"2025-04-10","arxiv_id":"2504.08161","repositories_listed":0,"syntology":null},{"url":null,"slug":"seed1-5-thinking-advancing-superb-reasoning","title":"Seed1.5-Thinking: Advancing Superb Reasoning Models with Reinforcement Learning","date":"2025-04-10","arxiv_id":"2504.13914","repositories_listed":0,"syntology":null},{"url":null,"slug":"better-decisions-through-the-right-causal","title":"Better Decisions through the Right Causal World Model","date":"2025-04-09","arxiv_id":"2504.07257","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-bounds-for-robust-online-decision","title":"Regret Bounds for Robust Online Decision Making","date":"2025-04-09","arxiv_id":"2504.06820","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-vehicle-routing-via-ai","title":"Accelerating Vehicle Routing via AI-Initialized Genetic Algorithms","date":"2025-04-08","arxiv_id":"2504.06126","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-exploration-in-reinforcement-learning","title":"Smart Exploration in Reinforcement Learning using Bounded Uncertainty Models","date":"2025-04-08","arxiv_id":"2504.05978","repositories_listed":0,"syntology":null},{"url":null,"slug":"tw-crl-time-weighted-contrastive-reward","title":"TW-CRL: Time-Weighted Contrastive Reward Learning for Efficient Inverse Reinforcement Learning","date":"2025-04-08","arxiv_id":"2504.05585","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithm-discovery-with-llms-evolutionary","title":"Algorithm Discovery With LLMs: Evolutionary Search Meets Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.05108","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-augmented-inverse-reinforcement","title":"Attention-Augmented Inverse Reinforcement Learning with Graph Convolutions for Multi-Agent Task Allocation","date":"2025-04-07","arxiv_id":"2504.05045","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensuring-safety-in-an-uncertain-environment","title":"Ensuring Safety in an Uncertain Environment: Constrained MDPs via Stochastic Thresholds","date":"2025-04-07","arxiv_id":"2504.04973","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-hierarchical-reinforcement-learning","title":"Federated Hierarchical Reinforcement Learning for Adaptive Traffic Signal Control","date":"2025-04-07","arxiv_id":"2504.05553","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyprl-reinforcement-learning-of-control","title":"HypRL: Reinforcement Learning of Control Policies for Hyperproperties","date":"2025-04-07","arxiv_id":"2504.04675","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-22","title":"Multi-Agent Deep Reinforcement Learning for Multiple Anesthetics Collaborative Control","date":"2025-04-07","arxiv_id":"2504.04765","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-data-generation-multi-step-rl-for","title":"Synthetic Data Generation & Multi-Step RL for Reasoning & Tool Use","date":"2025-04-07","arxiv_id":"2504.04736","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-role-of-environment-access-in-agnostic","title":"The Role of Environment Access in Agnostic Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.05405","repositories_listed":0,"syntology":null},{"url":null,"slug":"orbitzoo-multi-agent-reinforcement-learning","title":"OrbitZoo: Multi-Agent Reinforcement Learning Environment for Orbital Dynamics","date":"2025-04-05","arxiv_id":"2504.04160","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-penalty-based-bidirectional","title":"Enhanced Penalty-based Bidirectional Reinforcement Learning Algorithms","date":"2025-04-04","arxiv_id":"2504.03163","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-mixed-criticality-scheduling-with","title":"Improving Mixed-Criticality Scheduling with Reinforcement Learning","date":"2025-04-04","arxiv_id":"2504.03994","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dual-arm-coordination-for-grasping","title":"Learning Dual-Arm Coordination for Grasping Large Flat Objects","date":"2025-04-04","arxiv_id":"2504.03500","repositories_listed":0,"syntology":null},{"url":null,"slug":"moral-a-multimodal-reinforcement-learning","title":"MORAL: A Multimodal Reinforcement Learning Framework for Decision Making in Autonomous Laboratories","date":"2025-04-04","arxiv_id":"2504.03153","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-and-distributional-reinforcement-1","title":"Offline and Distributional Reinforcement Learning for Wireless Communications","date":"2025-04-04","arxiv_id":"2504.03804","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-difficulty-filtering-for-reasoning","title":"Online Difficulty Filtering for Reasoning Oriented Reinforcement Learning","date":"2025-04-04","arxiv_id":"2504.03380","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-quantum-circuits-via-zx-diagrams","title":"Optimizing Quantum Circuits via ZX Diagrams using Reinforcement Learning and Graph Neural Networks","date":"2025-04-04","arxiv_id":"2504.03429","repositories_listed":0,"syntology":null},{"url":null,"slug":"learnat-learning-nl2sql-with-ast-guided-task","title":"LearNAT: Learning NL2SQL with AST-guided Task Decomposition for Large Language Models","date":"2025-04-03","arxiv_id":"2504.02327","repositories_listed":0,"syntology":null},{"url":null,"slug":"symdqn-symbolic-knowledge-and-reasoning-in","title":"SymDQN: Symbolic Knowledge and Reasoning in Neural Network-based Reinforcement Learning","date":"2025-04-03","arxiv_id":"2504.02654","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-graph-reinforcement-learning-for-uav","title":"Deep Graph Reinforcement Learning for UAV-Enabled Multi-User Secure Communications","date":"2025-04-02","arxiv_id":"2504.01446","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpreting-emergent-planning-in-model-free","title":"Interpreting Emergent Planning in Model-Free Reinforcement Learning","date":"2025-04-02","arxiv_id":"2504.01871","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-curriculum-learning-for-goal","title":"Probabilistic Curriculum Learning for Goal-Based Reinforcement Learning","date":"2025-04-02","arxiv_id":"2504.01459","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-robust-dynamic","title":"Reinforcement learning for robust dynamic metabolic control","date":"2025-04-01","arxiv_id":"2504.00735","repositories_listed":0,"syntology":null},{"url":null,"slug":"expanding-rl-with-verifiable-rewards-across","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","date":"2025-03-31","arxiv_id":"2503.23829","repositories_listed":0,"syntology":null},{"url":null,"slug":"nuclear-microreactor-control-with-deep","title":"Nuclear Microreactor Control with Deep Reinforcement Learning","date":"2025-03-31","arxiv_id":"2504.00156","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-active-matter","title":"Reinforcement Learning for Active Matter","date":"2025-03-30","arxiv_id":"2503.23308","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-traffic-rule-compliance-using","title":"Predictive Traffic Rule Compliance using Reinforcement Learning","date":"2025-03-29","arxiv_id":"2503.22925","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl2grid-benchmarking-reinforcement-learning","title":"RL2Grid: Benchmarking Reinforcement Learning in Power Grid Operations","date":"2025-03-29","arxiv_id":"2503.23101","repositories_listed":0,"syntology":null},{"url":null,"slug":"crllk-constrained-reinforcement-learning-for","title":"CRLLK: Constrained Reinforcement Learning for Lane Keeping in Autonomous Driving","date":"2025-03-28","arxiv_id":"2503.22248","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-guided-sequence-weighting-for","title":"Entropy-guided sequence weighting for efficient exploration in RL-based LLM fine-tuning","date":"2025-03-28","arxiv_id":"2503.22456","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-efficient-and-1","title":"Reinforcement learning for efficient and robust multi-setpoint and multi-trajectory tracking in bioprocesses","date":"2025-03-28","arxiv_id":"2503.22409","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-machine-learning","title":"Reinforcement Learning for Machine Learning Model Deployment: Evaluating Multi-Armed Bandits in ML Ops Environments","date":"2025-03-28","arxiv_id":"2503.22595","repositories_listed":0,"syntology":null},{"url":null,"slug":"rldbf-enhancing-llms-via-reinforcement","title":"RLDBF: Enhancing LLMs Via Reinforcement Learning With DataBase FeedBack","date":"2025-03-28","arxiv_id":"2504.03713","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-offline-reinforcement-learning-3","title":"Model-Based Offline Reinforcement Learning with Adversarial Data Augmentation","date":"2025-03-26","arxiv_id":"2503.20285","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-discrete","title":"Offline Reinforcement Learning with Discrete Diffusion Skills","date":"2025-03-26","arxiv_id":"2503.20176","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-beyond-limits-advances-and-open","title":"Reasoning Beyond Limits: Advances and Open Problems for LLMs","date":"2025-03-26","arxiv_id":"2503.22732","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-aware-perturbation-optimization-for","title":"State-Aware Perturbation Optimization for Robust Deep Reinforcement Learning","date":"2025-03-26","arxiv_id":"2503.20613","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-world-models-for-bilevel","title":"Synthesizing world models for bilevel planning","date":"2025-03-26","arxiv_id":"2503.20124","repositories_listed":0,"syntology":null},{"url":null,"slug":"abstracting-geo-specific-terrains-to-scale-up","title":"Abstracting Geo-specific Terrains to Scale Up Reinforcement Learning","date":"2025-03-25","arxiv_id":"2503.20078","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-chain-of-thought-with-jensen-s","title":"Learning to chain-of-thought with Jensen's evidence lower bound","date":"2025-03-25","arxiv_id":"2503.19618","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-framework-to-rule-them-all-unifying-rl","title":"One Framework to Rule Them All: Unifying RL-Based and RL-Free Methods in RLHF","date":"2025-03-25","arxiv_id":"2503.19523","repositories_listed":0,"syntology":null},{"url":null,"slug":"adventurer-exploration-with-bigan-for-deep","title":"Adventurer: Exploration with BiGAN for Deep Reinforcement Learning","date":"2025-03-24","arxiv_id":"2503.18612","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-bounds-for-two-time-scale","title":"Finite-Time Bounds for Two-Time-Scale Stochastic Approximation with Arbitrary Norm Contractions and Markovian Noise","date":"2025-03-24","arxiv_id":"2503.18391","repositories_listed":0,"syntology":null},{"url":null,"slug":"option-discovery-using-llm-guided-semantic","title":"Option Discovery Using LLM-guided Semantic Hierarchical Reinforcement Learning","date":"2025-03-24","arxiv_id":"2503.19007","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-of-2","title":"Sample-Efficient Reinforcement Learning of Koopman eNMPC","date":"2025-03-24","arxiv_id":"2503.18787","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-multi-agent-reinforcement-learning","title":"Iterative Multi-Agent Reinforcement Learning: A Novel Approach Toward Real-World Multi-Echelon Inventory Optimization","date":"2025-03-23","arxiv_id":"2503.18201","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-sample-complexity-bounds-in-bilevel","title":"On The Sample Complexity Bounds In Bilevel Reinforcement Learning","date":"2025-03-22","arxiv_id":"2503.17644","repositories_listed":0,"syntology":null},{"url":null,"slug":"grammar-and-gameplay-aligned-rl-for-game","title":"Grammar and Gameplay-aligned RL for Game Description Generation with LLMs","date":"2025-03-20","arxiv_id":"2503.15783","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-automated-semantic-interpretability","title":"Towards Automated Semantic Interpretability in Reinforcement Learning via Vision-Language Models","date":"2025-03-20","arxiv_id":"2503.16724","repositories_listed":0,"syntology":null},{"url":null,"slug":"uas-visual-navigation-in-large-and-unseen","title":"UAS Visual Navigation in Large and Unseen Environments via a Meta Agent","date":"2025-03-20","arxiv_id":"2503.15781","repositories_listed":0,"syntology":null},{"url":null,"slug":"behaviour-discovery-and-attribution-for","title":"Behaviour Discovery and Attribution for Explainable Reinforcement Learning","date":"2025-03-19","arxiv_id":"2503.14973","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-review-of-reinforcement","title":"Comprehensive Review of Reinforcement Learning for Medical Ultrasound Imaging","date":"2025-03-19","arxiv_id":"2503.16543","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepmesh-auto-regressive-artist-mesh-creation","title":"DeepMesh: Auto-Regressive Artist-mesh Creation with Reinforcement Learning","date":"2025-03-19","arxiv_id":"2503.15265","repositories_listed":0,"syntology":null},{"url":null,"slug":"partially-observable-reinforcement-learning","title":"Partially Observable Reinforcement Learning with Memory Traces","date":"2025-03-19","arxiv_id":"2503.15200","repositories_listed":0,"syntology":null},{"url":null,"slug":"reachable-sets-based-trajectory-planning","title":"Reachable Sets-based Trajectory Planning Combining Reinforcement Learning and iLQR","date":"2025-03-19","arxiv_id":"2503.17398","repositories_listed":0,"syntology":null},{"url":null,"slug":"styleloco-generative-adversarial-distillation","title":"StyleLoco: Generative Adversarial Distillation for Natural Humanoid Robot Locomotion","date":"2025-03-19","arxiv_id":"2503.15082","repositories_listed":0,"syntology":null},{"url":null,"slug":"colson-controllable-learning-based-social","title":"COLSON: Controllable Learning-Based Social Navigation via Diffusion-Based Reinforcement Learning","date":"2025-03-18","arxiv_id":"2503.13934","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctsac-curriculum-based-transformer-soft-actor","title":"CTSAC: Curriculum-Based Transformer Soft Actor-Critic for Goal-Oriented Robot Exploration","date":"2025-03-18","arxiv_id":"2503.14254","repositories_listed":0,"syntology":null},{"url":null,"slug":"pauli-network-circuit-synthesis-with","title":"Pauli Network Circuit Synthesis with Reinforcement Learning","date":"2025-03-18","arxiv_id":"2503.14448","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-driven-transformer","title":"A Reinforcement Learning-Driven Transformer GAN for Molecular Generation","date":"2025-03-17","arxiv_id":"2503.12796","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifelong-reinforcement-learning-with-1","title":"Lifelong Reinforcement Learning with Similarity-Driven Weighting by Large Models","date":"2025-03-17","arxiv_id":"2503.12923","repositories_listed":0,"syntology":null}],"record_sha256":"279f7cacc336c7d419623d9d3c60cd7190ed6eb70d92c282ceb6e106ab65cb96","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}