{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/safe-reinforcement-learning/papers/2","list_of":"/task/safe-reinforcement-learning","task":"Safe Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":306,"counts":{"archive_papers_tagged":306,"with_a_code_link":100,"where_syntology_ran_a_sample":34,"not_listed_spam_title":0,"listed":306,"listed_where_code_ran":34,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":29,"every_run_a_failure_of_syntologys_instrument":5,"listed_with_a_run_with_no_instrument_failure":29,"listed_every_run_a_failure_of_syntologys_instrument":5,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/safe-reinforcement-learning","prev":"/task/safe-reinforcement-learning","next":"/task/safe-reinforcement-learning/papers/3","papers":[{"url":null,"slug":"provably-safe-reinforcement-learning-from","title":"Provably Safe Reinforcement Learning from Analytic Gradients","date":"2025-06-02","arxiv_id":"2506.01665","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-provable-approach-for-end-to-end-safe","title":"A Provable Approach for End-to-End Safe Reinforcement Learning","date":"2025-05-28","arxiv_id":"2505.21852","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-guarded-safe-reinforcement-learning","title":"Offline Guarded Safe Reinforcement Learning for Medical Treatment Optimization Strategies","date":"2025-05-22","arxiv_id":"2505.16242","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-aware-safe-reinforcement-learning-for","title":"Risk-Aware Safe Reinforcement Learning for Control of Stochastic Linear Systems","date":"2025-05-14","arxiv_id":"2505.09734","repositories_listed":0,"syntology":null},{"url":null,"slug":"feasibility-aware-pessimistic-estimation","title":"Feasibility-Aware Pessimistic Estimation: Toward Long-Horizon Safety in Offline RL","date":"2025-05-13","arxiv_id":"2505.08179","repositories_listed":0,"syntology":null},{"url":null,"slug":"skill-based-safe-reinforcement-learning-with","title":"Skill-based Safe Reinforcement Learning with Risk Planning","date":"2025-05-02","arxiv_id":"2505.01619","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-control-barrier-function-via","title":"Designing Control Barrier Function via Probabilistic Enumeration for Safe Reinforcement Learning Navigation","date":"2025-04-30","arxiv_id":"2504.21643","repositories_listed":0,"syntology":null},{"url":null,"slug":"anytime-safe-reinforcement-learning","title":"Anytime Safe Reinforcement Learning","date":"2025-04-23","arxiv_id":"2504.16417","repositories_listed":0,"syntology":null},{"url":null,"slug":"traces-trajectory-based-credit-assignment","title":"TraCeS: Trajectory Based Credit Assignment From Sparse Safety Feedback","date":"2025-04-17","arxiv_id":"2504.12557","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-natural-language-constraints-for","title":"Learning Natural Language Constraints for Safe Reinforcement Learning of Language Agents","date":"2025-04-04","arxiv_id":"2504.03185","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-modulation-enhancing-safety-in","title":"Safety Modulation: Enhancing Safety in Reinforcement Learning through Cost-Modulated Rewards","date":"2025-04-03","arxiv_id":"2504.03040","repositories_listed":0,"syntology":null},{"url":null,"slug":"bresa-bio-inspired-reflexive-safe","title":"Bresa: Bio-inspired Reflexive Safe Reinforcement Learning for Contact-Rich Robotic Tasks","date":"2025-03-27","arxiv_id":"2503.21989","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-rlhf-v-safe-reinforcement-learning-from","title":"Safe RLHF-V: Safe Reinforcement Learning from Human Feedback in Multimodal Large Language Models","date":"2025-03-22","arxiv_id":"2503.17682","repositories_listed":0,"syntology":null},{"url":null,"slug":"reachable-sets-based-trajectory-planning","title":"Reachable Sets-based Trajectory Planning Combining Reinforcement Learning and iLQR","date":"2025-03-19","arxiv_id":"2503.17398","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-safe","title":"Hierarchical Reinforcement Learning for Safe Mapless Navigation with Congestion Estimation","date":"2025-03-15","arxiv_id":"2503.12036","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhance-exploration-in-safe-reinforcement","title":"Enhance Exploration in Safe Reinforcement Learning with Contrastive Representation Learning","date":"2025-03-13","arxiv_id":"2503.10318","repositories_listed":0,"syntology":null},{"url":null,"slug":"hasard-a-benchmark-for-vision-based-safe","title":"HASARD: A Benchmark for Vision-Based Safe Reinforcement Learning in Embodied Agents","date":"2025-03-11","arxiv_id":"2503.08241","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-shielding-for-safe","title":"Probabilistic Shielding for Safe Reinforcement Learning","date":"2025-03-09","arxiv_id":"2503.07671","repositories_listed":0,"syntology":null},{"url":null,"slug":"safevla-towards-safety-alignment-of-vision","title":"SafeVLA: Towards Safety Alignment of Vision-Language-Action Model via Constrained Learning","date":"2025-03-05","arxiv_id":"2503.03480","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-explore-when-mistakes-are-not","title":"Learning to explore when mistakes are not allowed","date":"2025-02-19","arxiv_id":"2502.13801","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-based-control-for","title":"Safe Reinforcement Learning-based Control for Hydrogen Diesel Dual-Fuel Engines","date":"2025-02-13","arxiv_id":"2502.09826","repositories_listed":0,"syntology":null},{"url":null,"slug":"dial-distribution-informed-adaptive-learning","title":"DIAL: Distribution-Informed Adaptive Learning of Multi-Task Constraints for Safety-Critical Systems","date":"2025-01-30","arxiv_id":"2501.18086","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-real-world","title":"Safe Reinforcement Learning for Real-World Engine Control","date":"2025-01-28","arxiv_id":"2501.16613","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-minimal","title":"Safe Reinforcement Learning with Minimal Supervision","date":"2025-01-08","arxiv_id":"2501.04481","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-multiagent-coordination-via-entropic","title":"Safe Multiagent Coordination via Entropic Exploration","date":"2024-12-29","arxiv_id":"2412.20361","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-model-guided-worst-case-sampling-for","title":"Physics-model-guided Worst-case Sampling for Safe Reinforcement Learning","date":"2024-12-17","arxiv_id":"2412.13224","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-using-finite","title":"Safe Reinforcement Learning using Finite-Horizon Gradient-based Estimation","date":"2024-12-15","arxiv_id":"2412.11138","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-text-to-trajectory-exploring-complex","title":"From Text to Trajectory: Exploring Complex Constraint Representation and Decomposition in Safe Reinforcement Learning","date":"2024-12-12","arxiv_id":"2412.08920","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl2-reinforce-large-language-model-to-assist","title":"RL2: Reinforce Large Language Model to Assist Safe Reinforcement Learning for Energy Management of Active Distribution Networks","date":"2024-12-02","arxiv_id":"2412.01303","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensuring-safety-in-target-pursuit-control-a","title":"Ensuring Safety in Target Pursuit Control: A CBF-Safe Reinforcement Learning Approach","date":"2024-11-26","arxiv_id":"2411.17552","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-safeguards-for-safe-and-model","title":"Progressive Safeguards for Safe and Model-Agnostic Reinforcement Learning","date":"2024-10-31","arxiv_id":"2410.24096","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmented-lagrangian-based-safe-reinforcement","title":"Augmented Lagrangian-Based Safe Reinforcement Learning Approach for Distribution System Volt/VAR Control","date":"2024-10-19","arxiv_id":"2410.15188","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-regret-bound-for-safe-reinforcement","title":"Improved Regret Bound for Safe Reinforcement Learning via Tighter Cost Pessimism and Reward Optimism","date":"2024-10-14","arxiv_id":"2410.10158","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-safety-modulator-actor-critic-method-in","title":"A Safety Modulator Actor-Critic Method in Model-Free Safe Reinforcement Learning and Application in UAV Hovering","date":"2024-10-09","arxiv_id":"2410.06847","repositories_listed":0,"syntology":null},{"url":null,"slug":"flipping-based-policy-for-chance-constrained","title":"Flipping-based Policy for Chance-Constrained Markov Decision Processes","date":"2024-10-09","arxiv_id":"2410.06474","repositories_listed":0,"syntology":null},{"url":null,"slug":"realizable-continuous-space-shields-for-safe","title":"Realizable Continuous-Space Shields for Safe Reinforcement Learning","date":"2024-10-02","arxiv_id":"2410.02038","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-critical-review-of-safe-reinforcement","title":"A Critical Review of Safe Reinforcement Learning Techniques in Smart Grid Applications","date":"2024-09-24","arxiv_id":"2409.16256","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-oriented-pruning-and-interpretation-of","title":"Safety-Oriented Pruning and Interpretation of Reinforcement Learning Policies","date":"2024-09-16","arxiv_id":"2409.10218","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-management-of-grid-interactive","title":"Optimal Management of Grid-Interactive Efficient Buildings via Safe Reinforcement Learning","date":"2024-09-12","arxiv_id":"2409.08132","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-safe-exploration-in-safe","title":"Revisiting Safe Exploration in Safe Reinforcement learning","date":"2024-09-02","arxiv_id":"2409.01245","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-learning-to-plan","title":"Bridging the gap between Learning-to-plan, Motion Primitives and Safe Reinforcement Learning","date":"2024-08-26","arxiv_id":"2408.14063","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-sac-lag-towards-deployable-safe","title":"Meta SAC-Lag: Towards Deployable Safe Reinforcement Learning via MetaGradient-based Hyperparameter Tuning","date":"2024-08-15","arxiv_id":"2408.07962","repositories_listed":0,"syntology":null},{"url":null,"slug":"anomalous-state-sequence-modeling-to-enhance","title":"Anomalous State Sequence Modeling to Enhance Safety in Reinforcement Learning","date":"2024-07-29","arxiv_id":"2407.19860","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-sustainable-energy","title":"Reinforcement Learning for Sustainable Energy: A Survey","date":"2024-07-26","arxiv_id":"2407.18597","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathrm-e-2-cfd-towards-effective-and","title":"$\\mathrm{E^{2}CFD}$: Towards Effective and Efficient Cost Function Design for Safe Reinforcement Learning via Large Language Model","date":"2024-07-08","arxiv_id":"2407.05580","repositories_listed":0,"syntology":null},{"url":null,"slug":"fosp-fine-tuning-offline-safe-policy-through","title":"FOSP: Fine-tuning Offline Safe Policy through World Models","date":"2024-07-06","arxiv_id":"2407.04942","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-cor-a-dual-expert-approach-to","title":"Safe CoR: A Dual-Expert Approach to Integrating Imitation Learning and Safe Reinforcement Learning Using Constraint Rewards","date":"2024-07-02","arxiv_id":"2407.02245","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-power-system","title":"Safe Reinforcement Learning for Power System Control: A Review","date":"2024-06-30","arxiv_id":"2407.00681","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-safe-reinforcement-learning-1","title":"A Review of Safe Reinforcement Learning Methods for Modern Power Systems","date":"2024-06-29","arxiv_id":"2407.00304","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-safe-reinforcement-learning-enabled","title":"Adaptive Safe Reinforcement Learning-Enabled Optimization of Battery Fast-Charging Protocols","date":"2024-06-18","arxiv_id":"2406.12309","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-transport-assisted-risk-sensitive-q","title":"Optimal Transport-Assisted Risk-Sensitive Q-Learning","date":"2024-06-17","arxiv_id":"2406.11774","repositories_listed":0,"syntology":null},{"url":null,"slug":"cimrl-combining-imitiation-and-reinforcement","title":"CIMRL: Combining IMitation and Reinforcement Learning for Safe Autonomous Driving","date":"2024-06-13","arxiv_id":"2406.08878","repositories_listed":0,"syntology":null},{"url":null,"slug":"gensafe-a-generalizable-safety-enhancer-for","title":"GenSafe: A Generalizable Safety Enhancer for Safe Reinforcement Learning Algorithms Based on Reduced Order Markov Decision Process Model","date":"2024-06-06","arxiv_id":"2406.03912","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-through-permissibility-shield","title":"Safety through Permissibility: Shield Construction for Fast and Safe Reinforcement Learning","date":"2024-05-29","arxiv_id":"2405.19414","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-risk-safe-reinforcement-learning","title":"Spectral-Risk Safe Reinforcement Learning with Convergence Guarantees","date":"2024-05-29","arxiv_id":"2405.18698","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cmdp-within-online-framework-for-meta-safe","title":"A CMDP-within-online framework for Meta-Safe Reinforcement Learning","date":"2024-05-26","arxiv_id":"2405.16601","repositories_listed":0,"syntology":null},{"url":null,"slug":"make-safe-decisions-in-power-system-safe","title":"Make Safe Decisions in Power System: Safe Reinforcement Learning Based Pre-decision Making for Voltage Stability Emergency Control","date":"2024-05-26","arxiv_id":"2405.16485","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-model-predictive-shielding-for","title":"Dynamic Model Predictive Shielding for Provably Safe Reinforcement Learning","date":"2024-05-22","arxiv_id":"2405.13863","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-no-harm-a-counterfactual-approach-to-safe","title":"Do No Harm: A Counterfactual Approach to Safe Reinforcement Learning","date":"2024-05-19","arxiv_id":"2405.11669","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-learned-non","title":"Safe Reinforcement Learning with Learned Non-Markovian Safety Constraints","date":"2024-05-05","arxiv_id":"2405.03005","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-safe-set-algorithm-for-provably-safe","title":"Implicit Safe Set Algorithm for Provably Safe Reinforcement Learning","date":"2024-05-04","arxiv_id":"2405.02754","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-control-barrier-functions-and-their","title":"Learning Control Barrier Functions and their application in Reinforcement Learning: A Survey","date":"2024-04-22","arxiv_id":"2404.16879","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-on-the-constraint","title":"Safe Reinforcement Learning on the Constraint Manifold: Theory and Applications","date":"2024-04-13","arxiv_id":"2404.09080","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-and-short-term-constraints-driven-safe","title":"Long and Short-Term Constraints Driven Safe Reinforcement Learning for Autonomous Driving","date":"2024-03-27","arxiv_id":"2403.18209","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-constrained","title":"Safe Reinforcement Learning for Constrained Markov Decision Processes with Stochastic Stopping Time","date":"2024-03-23","arxiv_id":"2403.15928","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-logic-specification-conditioned","title":"Temporal Logic Specification-Conditioned Decision Transformer for Offline Safe Reinforcement Learning","date":"2024-02-27","arxiv_id":"2402.17217","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-constraint-safe-rl-with-objective","title":"Uniformly Safe RL with Objective Suppression for Multi-Constraint Safety-Critical Applications","date":"2024-02-23","arxiv_id":"2402.15650","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-optimized-reinforcement-learning-via","title":"Safety Optimized Reinforcement Learning via Multi-Objective Policy Optimization","date":"2024-02-23","arxiv_id":"2402.15197","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-traffic-rule-compliance-in-safe","title":"Provable Traffic Rule Compliance in Safe Reinforcement Learning on the Open Sea","date":"2024-02-13","arxiv_id":"2402.08502","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-network-constrained-operational","title":"Techno-Economic Modeling and Safe Operational Optimization of Multi-Network Constrained Integrated Community Energy Systems","date":"2024-02-08","arxiv_id":"2402.05412","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-safe-reinforcement-learning-driven-weights","title":"A Safe Reinforcement Learning driven Weights-varying Model Predictive Control for Autonomous Vehicle Motion Control","date":"2024-02-04","arxiv_id":"2402.02624","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-constraint-formulations-in-safe","title":"A Survey of Constraint Formulations in Safe Reinforcement Learning","date":"2024-02-03","arxiv_id":"2402.02025","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-primal-dual-method-for-safe","title":"Adaptive Primal-Dual Method for Safe Reinforcement Learning","date":"2024-02-01","arxiv_id":"2402.00355","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-based-eco-driving","title":"Safe Reinforcement Learning-Based Eco-Driving Control for Mixed Traffic Flows With Disturbances","date":"2024-01-31","arxiv_id":"2401.17837","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-safe-reinforcement-learning-algorithm-for","title":"A Safe Reinforcement Learning Algorithm for Supervisory Control of Power Plants","date":"2024-01-23","arxiv_id":"2401.13020","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-free-form","title":"Safe Reinforcement Learning with Free-form Natural Language Constraints and Pre-Trained Language Models","date":"2024-01-15","arxiv_id":"2401.07553","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-term-safe-reinforcement-learning-with","title":"Long-term Safe Reinforcement Learning with Binary Feedback","date":"2024-01-08","arxiv_id":"2401.03786","repositories_listed":0,"syntology":null},{"url":null,"slug":"gradient-shaping-for-multi-constraint-safe","title":"Gradient Shaping for Multi-Constraint Safe Reinforcement Learning","date":"2023-12-23","arxiv_id":"2312.15127","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-1","title":"Safe Reinforcement Learning with Instantaneous Constraints: The Role of Aggressive Exploration","date":"2023-12-22","arxiv_id":"2312.14470","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-exploration-in-reinforcement-learning","title":"Safe Exploration in Reinforcement Learning: Training Backup Control Barrier Functions with Zero Training Time Safety Violations","date":"2023-12-13","arxiv_id":"2312.07828","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-risk-in-reinforcement-learning-a","title":"Modeling Risk in Reinforcement Learning: A Literature Mapping","date":"2023-12-08","arxiv_id":"2312.05231","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-off-policy-safe-reinforcement","title":"Efficient Off-Policy Safe Reinforcement Learning Using Trust Region Conditional Value at Risk","date":"2023-12-01","arxiv_id":"2312.00342","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-in-tensor","title":"Safe Reinforcement Learning in Tensor Reproducing Kernel Hilbert Space","date":"2023-12-01","arxiv_id":"2312.00727","repositories_listed":0,"syntology":null},{"url":null,"slug":"trc-trust-region-conditional-value-at-risk","title":"TRC: Trust Region Conditional Value at Risk for Safe Reinforcement Learning","date":"2023-12-01","arxiv_id":"2312.00344","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-in-a-simulated","title":"Safe Reinforcement Learning in a Simulated Robotic Arm","date":"2023-11-28","arxiv_id":"2312.09468","repositories_listed":0,"syntology":null},{"url":null,"slug":"networked-multiagent-safe-reinforcement","title":"Networked Multiagent Safe Reinforcement Learning for Low-carbon Demand Management in Distribution Network","date":"2023-11-27","arxiv_id":"2311.15594","repositories_listed":0,"syntology":null},{"url":null,"slug":"scheduling-distributed-flexible-assembly","title":"Scheduling Distributed Flexible Assembly Lines using Safe Reinforcement Learning with Soft Shielding","date":"2023-11-21","arxiv_id":"2311.12572","repositories_listed":0,"syntology":null},{"url":null,"slug":"scpo-safe-reinforcement-learning-with-safety","title":"SCPO: Safe Reinforcement Learning with Safety Critic Policy Optimization","date":"2023-11-01","arxiv_id":"2311.00880","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-gymnasium-a-unified-safe-reinforcement","title":"Safety-Gymnasium: A Unified Safe Reinforcement Learning Benchmark","date":"2023-10-19","arxiv_id":"2310.12567","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-safe-reinforcement-learning-under","title":"Robust Safe Reinforcement Learning under Adversarial Disturbances","date":"2023-10-11","arxiv_id":"2310.07207","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-a-safety-embedded","title":"Reinforcement Learning in a Safety-Embedded MDP with Trajectory Optimization","date":"2023-10-10","arxiv_id":"2310.06903","repositories_listed":0,"syntology":null},{"url":null,"slug":"constraint-conditioned-policy-optimization","title":"Constraint-Conditioned Policy Optimization for Versatile Safe Reinforcement Learning","date":"2023-10-05","arxiv_id":"2310.03718","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributionally-safe-reinforcement-learning","title":"Distributionally Safe Reinforcement Learning under Model Uncertainty: A Single-Level Approach by Differentiable Convex Programming","date":"2023-10-03","arxiv_id":"2310.02459","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-inhibitory-control-for-safe","title":"Risk-Sensitive Inhibitory Control for Safe Reinforcement Learning","date":"2023-10-02","arxiv_id":"2310.01538","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-reachability-estimation-for-safe","title":"Iterative Reachability Estimation for Safe Reinforcement Learning","date":"2023-09-24","arxiv_id":"2309.13528","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-recover-for-safe-reinforcement","title":"Learning to Recover for Safe Reinforcement Learning","date":"2023-09-21","arxiv_id":"2309.11907","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-online-distillation-promoting-safe","title":"Guided Online Distillation: Promoting Safe Reinforcement Learning by Offline Demonstration","date":"2023-09-18","arxiv_id":"2309.09408","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-dual","title":"Safe Reinforcement Learning with Dual Robustness","date":"2023-09-13","arxiv_id":"2309.06835","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-reward-structures-of-markov-decision","title":"On Reward Structures of Markov Decision Processes","date":"2023-08-28","arxiv_id":"2308.14919","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-optimal-head-to-head-autonomous","title":"Towards Optimal Head-to-head Autonomous Racing with Curriculum Reinforcement Learning","date":"2023-08-25","arxiv_id":"2308.13491","repositories_listed":0,"syntology":null}],"record_sha256":"ef456ee1c832128adc70c6794834909f5259ab30eb8bc6ab3a2bbd365e77fadd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}