{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/safe-reinforcement-learning/papers/3","list_of":"/task/safe-reinforcement-learning","task":"Safe Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":4,"rows_per_page":100,"rows":[201,300],"of":306,"counts":{"archive_papers_tagged":306,"with_a_code_link":100,"where_syntology_ran_a_sample":34,"not_listed_spam_title":0,"listed":306,"listed_where_code_ran":34,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":29,"every_run_a_failure_of_syntologys_instrument":5,"listed_with_a_run_with_no_instrument_failure":29,"listed_every_run_a_failure_of_syntologys_instrument":5,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/safe-reinforcement-learning","prev":"/task/safe-reinforcement-learning/papers/2","next":"/task/safe-reinforcement-learning/papers/4","papers":[{"url":null,"slug":"safeguarding-learning-based-control-for-smart","title":"Safeguarding Learning-based Control for Smart Energy Systems with Sampling Specifications","date":"2023-08-11","arxiv_id":"2308.06069","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-strategic","title":"Safe Reinforcement Learning for Strategic Bidding of Virtual Power Plants in Day-Ahead Markets","date":"2023-07-11","arxiv_id":"2307.05812","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-aware-task-composition-for-discrete","title":"Safety-Aware Task Composition for Discrete and Continuous Reinforcement Learning","date":"2023-06-29","arxiv_id":"2306.17033","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-dead-ends","title":"Safe Reinforcement Learning with Dead-Ends Avoidance and Recovery","date":"2023-06-24","arxiv_id":"2306.13944","repositories_listed":0,"syntology":null},{"url":null,"slug":"cancellation-free-regret-bounds-for","title":"Cancellation-Free Regret Bounds for Lagrangian Approaches in Constrained Markov Decision Processes","date":"2023-06-12","arxiv_id":"2306.07001","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-generalized-lagrangian","title":"Provably Efficient Generalized Lagrangian Policy Optimization for Safe Multi-Agent Reinforcement Learning","date":"2023-05-31","arxiv_id":"2306.00212","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-invariant-set-enhanced-safe","title":"Control invariant set enhanced safe reinforcement learning: improved sampling efficiency, guaranteed stability and robustness","date":"2023-05-24","arxiv_id":"2305.15602","repositories_listed":0,"syntology":null},{"url":null,"slug":"hjb-based-online-safe-reinforcement-learning","title":"Lagrangian-based online safe reinforcement learning for state-constrained systems","date":"2023-05-22","arxiv_id":"2305.12967","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-learning-of-policy-with-unknown","title":"Joint Learning of Policy with Unknown Temporal Constraints for Safe Reinforcement Learning","date":"2023-04-30","arxiv_id":"2305.00576","repositories_listed":0,"syntology":null},{"url":null,"slug":"feasible-policy-iteration","title":"Feasible Policy Iteration for Safe Reinforcement Learning","date":"2023-04-18","arxiv_id":"2304.08845","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-self","title":"An adaptive safety layer with hard constraints for safe reinforcement learning in multi-energy management systems","date":"2023-04-18","arxiv_id":"2304.08897","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-bellman-s-principle-of-optimality-and","title":"On Bellman's principle of optimality and Reinforcement learning for safety-constrained Markov decision process","date":"2023-02-25","arxiv_id":"2302.13152","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-safe-reinforcement-learning-with","title":"Provably Safe Reinforcement Learning with Step-wise Violation Constraints","date":"2023-02-13","arxiv_id":"2302.06064","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-near-optimal-algorithm-for-safe","title":"A Near-Optimal Algorithm for Safe Reinforcement Learning Under Instantaneous Hard Constraints","date":"2023-02-08","arxiv_id":"2302.04375","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-aggregation-for-safety-critical","title":"Adaptive Aggregation for Safety-Critical Control","date":"2023-02-07","arxiv_id":"2302.03586","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-wise-safe-reinforcement-learning-a","title":"State-wise Safe Reinforcement Learning: A Survey","date":"2023-02-06","arxiv_id":"2302.03122","repositories_listed":0,"syntology":null},{"url":null,"slug":"saformer-a-conditional-sequence-modeling","title":"SaFormer: A Conditional Sequence Modeling Approach to Offline Safe Reinforcement Learning","date":"2023-01-28","arxiv_id":"2301.12203","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-an-energy","title":"Safe Reinforcement Learning for an Energy-Efficient Driver Assistance System","date":"2023-01-03","arxiv_id":"2301.00904","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-causal-temporal-reasoning-for-markov","title":"Causal Temporal Reasoning for Markov Decision Processes","date":"2022-12-16","arxiv_id":"2212.08712","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-correction-from-baseline-towards-the","title":"Safety Correction from Baseline: Towards the Risk-aware Policy in Robotics via Dual-agent Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.06998","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with","title":"Safe Reinforcement Learning with Probabilistic Control Barrier Functions for Ramp Merging","date":"2022-12-01","arxiv_id":"2212.00618","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-model-free-reinforcement-learning-using","title":"Safe and Efficient Reinforcement Learning Using Disturbance-Observer-Based Control Barrier Functions","date":"2022-11-30","arxiv_id":"2211.17250","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-using-data-driven","title":"Safe Reinforcement Learning using Data-Driven Predictive Control","date":"2022-11-20","arxiv_id":"2211.11027","repositories_listed":0,"syntology":null},{"url":null,"slug":"lmpriors-pre-trained-language-models-as-task","title":"LMPriors: Pre-Trained Language Models as Task-Specific Priors","date":"2022-10-22","arxiv_id":"2210.12530","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-safe-reinforcement-learning-via","title":"Provably Safe Reinforcement Learning via Action Projection using Reachability Analysis and Polynomial Zonotopes","date":"2022-10-19","arxiv_id":"2210.10691","repositories_listed":0,"syntology":null},{"url":null,"slug":"enforcing-hard-constraints-with-soft-barriers","title":"Enforcing Hard Constraints with Soft Barriers: Safe Reinforcement Learning in Unknown Stochastic Environments","date":"2022-09-29","arxiv_id":"2209.15090","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-of-dynamic-high","title":"Safe Reinforcement Learning of Dynamic High-Dimensional Robotic Tasks: Navigation, Manipulation, Interaction","date":"2022-09-27","arxiv_id":"2209.13308","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-control-for","title":"Safe reinforcement learning control for continuous-time nonlinear systems without a backup controller","date":"2022-09-19","arxiv_id":"2209.08922","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-contrastive","title":"Safe Reinforcement Learning with Contrastive Risk Prediction","date":"2022-09-10","arxiv_id":"2209.09648","repositories_listed":0,"syntology":null},{"url":null,"slug":"rasr-risk-averse-soft-robust-mdps-with-evar","title":"RASR: Risk-Averse Soft-Robust MDPs with EVaR and Entropic Risk","date":"2022-09-09","arxiv_id":"2209.04067","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-contact-safe-reinforcement-learning","title":"A Contact-Safe Reinforcement Learning Framework for Contact-Rich Robot Manipulation","date":"2022-07-27","arxiv_id":"2207.13438","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-action-governor-for-uncertain","title":"Robust Action Governor for Uncertain Piecewise Affine Systems with Non-convex Constraints and Safe Reinforcement Learning","date":"2022-07-17","arxiv_id":"2207.08240","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-near-optimal-primal-dual-method-for-off","title":"A Near-Optimal Primal-Dual Method for Off-Policy Learning in CMDP","date":"2022-07-13","arxiv_id":"2207.06147","repositories_listed":0,"syntology":null},{"url":null,"slug":"ablation-study-of-how-run-time-assurance","title":"Ablation Study of How Run Time Assurance Impacts the Training and Performance of Reinforcement Learning Agents","date":"2022-07-08","arxiv_id":"2207.04117","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-multi-energy","title":"Safe reinforcement learning for multi-energy management systems with known constraint functions","date":"2022-07-08","arxiv_id":"2207.03830","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-via-confidence","title":"Safe Reinforcement Learning via Confidence-Based Filters","date":"2022-07-04","arxiv_id":"2207.01337","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-a-robot-being","title":"Safe Reinforcement Learning for a Robot Being Pursued but with Objectives Covering More Than Capture-avoidance","date":"2022-07-02","arxiv_id":"2207.00842","repositories_listed":0,"syntology":null},{"url":null,"slug":"penalized-proximal-policy-optimization-for","title":"Penalized Proximal Policy Optimization for Safe Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.11814","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-safe-reinforcement-learning-a","title":"Provably Safe Reinforcement Learning: Conceptual Analysis, Survey, and Benchmarking","date":"2022-05-13","arxiv_id":"2205.06750","repositories_listed":0,"syntology":null},{"url":null,"slug":"contingency-constrained-economic-dispatch","title":"Contingency-constrained economic dispatch with safe reinforcement learning","date":"2022-05-12","arxiv_id":"2205.06212","repositories_listed":0,"syntology":null},{"url":null,"slug":"saac-safe-reinforcement-learning-as-an","title":"SAAC: Safe Reinforcement Learning as an Adversarial Game of Actor-Critics","date":"2022-04-20","arxiv_id":"2204.09424","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-via-shielding-for","title":"Safe Reinforcement Learning via Shielding under Partial Observability","date":"2022-04-02","arxiv_id":"2204.00755","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-legged","title":"Safe Reinforcement Learning for Legged Locomotion","date":"2022-03-05","arxiv_id":"2203.02638","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-a-shield-from-catastrophic-action","title":"Learning a Shield from Catastrophic Action Effects: Never Repeat the Same Mistake","date":"2022-02-19","arxiv_id":"2202.09516","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-safe-reinforcement-learning-for","title":"Multi-task Safe Reinforcement Learning for Navigating Intersections in Dense Traffic","date":"2022-02-19","arxiv_id":"2202.09644","repositories_listed":0,"syntology":null},{"url":null,"slug":"safer-data-efficient-and-safe-reinforcement-1","title":"SAFER: Data-Efficient and Safe Reinforcement Learning via Skill Acquisition","date":"2022-02-10","arxiv_id":"2202.04849","repositories_listed":0,"syntology":null},{"url":null,"slug":"coordinated-frequency-control-through-safe","title":"Coordinated Frequency Control through Safe Reinforcement Learning","date":"2022-01-30","arxiv_id":"2202.00530","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-lab-to-real-safe-reinforcement","title":"Sim-to-Lab-to-Real: Safe Reinforcement Learning with Shielding and Generalization Guarantees","date":"2022-01-20","arxiv_id":"2201.08355","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-chance","title":"Safe Reinforcement Learning with Chance-constrained Model Predictive Control","date":"2021-12-27","arxiv_id":"2112.13941","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-multi-agent-deep-reinforcement-learning","title":"Safe multi-agent deep reinforcement learning for joint bidding and maintenance scheduling of generation units","date":"2021-12-20","arxiv_id":"2112.10459","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-safe-reinforcement-learning-with","title":"Model-Based Safe Reinforcement Learning with Time-Varying State and Control Constraints: An Application to Intelligent Vehicles","date":"2021-12-18","arxiv_id":"2112.11217","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-safe-deep-reinforcement-learning","title":"Benchmarking Safe Deep Reinforcement Learning in Aquatic Navigation","date":"2021-12-16","arxiv_id":"2112.10593","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-grid-voltage","title":"Safe Reinforcement Learning for Grid Voltage Control","date":"2021-12-02","arxiv_id":"2112.01484","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-be-cautious","title":"Learning to Be Cautious","date":"2021-10-29","arxiv_id":"2110.15907","repositories_listed":0,"syntology":null},{"url":null,"slug":"desta-a-framework-for-safe-reinforcement-1","title":"DESTA: A Framework for Safe Reinforcement Learning with Markov Games of Intervention","date":"2021-10-27","arxiv_id":"2110.14468","repositories_listed":0,"syntology":null},{"url":null,"slug":"computationally-efficient-safe-reinforcement","title":"Computationally Efficient Safe Reinforcement Learning for Power Systems","date":"2021-10-20","arxiv_id":"2110.10333","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-aware-policy-optimisation-for-1","title":"Safe Autonomous Racing via Approximate Reachability on Ego-vision","date":"2021-10-14","arxiv_id":"2110.07699","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-safe-reinforcement-learning-for","title":"Decentralized Safe Reinforcement Learning for Voltage Control","date":"2021-10-03","arxiv_id":"2110.01126","repositories_listed":0,"syntology":null},{"url":null,"slug":"safer-data-efficient-and-safe-reinforcement","title":"SAFER: Data-Efficient and Safe Reinforcement Learning Through Skill Acquisition","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"data-generation-method-for-learning-a-low","title":"Data Generation Method for Learning a Low-dimensional Safe Region in Safe Reinforcement Learning","date":"2021-09-10","arxiv_id":"2109.05077","repositories_listed":0,"syntology":null},{"url":null,"slug":"lyapunov-based-uncertainty-aware-safe","title":"Lyapunov-based uncertainty-aware safe reinforcement learning","date":"2021-07-29","arxiv_id":"2107.13944","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-linear","title":"Safe Reinforcement Learning with Linear Function Approximation","date":"2021-06-11","arxiv_id":"2106.06239","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-logic-based-intermittent-optimal-and","title":"Temporal-Logic-Based Intermittent, Optimal, and Safe Continuous-Time Learning for Trajectory Tracking","date":"2021-04-06","arxiv_id":"2104.02547","repositories_listed":0,"syntology":null},{"url":null,"slug":"barrier-function-based-safe-reinforcement","title":"Barrier Function-based Safe Reinforcement Learning for Emergency Control of Power Systems","date":"2021-03-26","arxiv_id":"2103.14186","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-exploration-process-adjustment-for","title":"Automatic Exploration Process Adjustment for Safe Reinforcement Learning with Joint Chance Constraint Satisfaction","date":"2021-03-05","arxiv_id":"2103.03656","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-using-robust","title":"Safe Reinforcement Learning Using Robust Action Governor","date":"2021-02-21","arxiv_id":"2102.10643","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-properties-of-kullback-leibler","title":"On the Properties of Kullback-Leibler Divergence Between Multivariate Gaussian Distributions","date":"2021-02-10","arxiv_id":"2102.05485","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-learning-and-optimization-techniques","title":"Safe Learning and Optimization Techniques: Towards a Survey of the State of the Art","date":"2021-01-23","arxiv_id":"2101.09505","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-safe-reinforcement-learning-for","title":"Context-Aware Safe Reinforcement Learning for Non-Stationary Environments","date":"2021-01-02","arxiv_id":"2101.00531","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-stability","title":"Learning for MPC with Stability & Safety Guarantees","date":"2020-12-14","arxiv_id":"2012.07369","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-antenna-tilt","title":"A Safe Reinforcement Learning Architecture for Antenna Tilt Optimisation","date":"2020-12-02","arxiv_id":"2012.01296","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-mixture-weights-for-off-policy","title":"Optimal Mixture Weights for Off-Policy Evaluation with Multiple Behavior Policies","date":"2020-11-29","arxiv_id":"2011.14359","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-primal-approach-to-constrained-policy-1","title":"CRPO: A New Approach for Safe Reinforcement Learning with Convergence Guarantee","date":"2020-11-11","arxiv_id":"2011.05869","repositories_listed":0,"syntology":null},{"url":null,"slug":"remote-electrical-tilt-optimization-via-safe","title":"Remote Electrical Tilt Optimization via Safe Reinforcement Learning","date":"2020-10-12","arxiv_id":"2010.05842","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-natural-1","title":"Safe Reinforcement Learning with Natural Language Constraints","date":"2020-10-11","arxiv_id":"2010.05150","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-primal-approach-to-constrained-policy","title":"A Primal Approach to Constrained Policy Optimization: Global Optimality and Finite-Time Analysis","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-safety-aware-model-based-reinforcement","title":"Safe Model-Based Reinforcement Learning for Systems with Parametric Uncertainties","date":"2020-07-24","arxiv_id":"2007.12666","repositories_listed":0,"syntology":null},{"url":null,"slug":"responsive-safety-in-reinforcement-learning","title":"Responsive Safety in Reinforcement Learning by PID Lagrangian Methods","date":"2020-07-08","arxiv_id":"2007.03964","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-mixture","title":"Safe Reinforcement Learning with Mixture Density Network: A Case Study in Autonomous Highway Driving","date":"2020-07-02","arxiv_id":"2007.01698","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-safe-reinforcement-learning-with","title":"Accelerating Safe Reinforcement Learning with Constraint-mismatched Policies","date":"2020-06-20","arxiv_id":"2006.11645","repositories_listed":0,"syntology":null},{"url":null,"slug":"set-invariant-constrained-reinforcement","title":"FISAR: Forward Invariant Safe Reinforcement Learning with a Deep Neural Network-Based Optimize","date":"2020-06-19","arxiv_id":"2006.11419","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-through-meta","title":"Safe Reinforcement Learning through Meta-learned Instincts","date":"2020-05-06","arxiv_id":"2005.03233","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-via-projection-on","title":"Safe Reinforcement Learning via Projection on a Safe Set: How to Achieve Optimality?","date":"2020-04-02","arxiv_id":"2004.00915","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-autonomous","title":"Safe Reinforcement Learning for Autonomous Vehicles through Parallel Constrained Policy Optimization","date":"2020-03-03","arxiv_id":"2003.01303","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-safe-exploration-via","title":"Provably Efficient Safe Exploration via Primal-Dual Policy Optimization","date":"2020-03-01","arxiv_id":"2003.00534","repositories_listed":0,"syntology":null},{"url":null,"slug":"responsive-safety-in-reinforcement-learning-1","title":"Responsive Safety in Reinforcement Learning","date":"2020-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"doubly-robust-off-policy-actor-critic","title":"Doubly Robust Off-Policy Actor-Critic Algorithms for Reinforcement Learning","date":"2019-12-11","arxiv_id":"1912.05109","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-guarantees-for-planning-based-on","title":"Safety Guarantees for Planning Based on Iterative Gaussian Processes","date":"2019-11-29","arxiv_id":"1912.00071","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-bayesian-recurrent-neural-networks-for","title":"Fully Bayesian Recurrent Neural Networks for Safe Reinforcement Learning","date":"2019-11-08","arxiv_id":"1911.03308","repositories_listed":0,"syntology":null},{"url":null,"slug":"case-study-verifying-the-safety-of-an","title":"Case Study: Verifying the Safety of an Autonomous Racing Car with a Neural Network Controller","date":"2019-10-24","arxiv_id":"1910.11309","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-model-predictive-shielding-for-safe","title":"Robust Model Predictive Shielding for Safe Reinforcement Learning with Stochastic Dynamics","date":"2019-10-24","arxiv_id":"1910.10885","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-driving-using-safe-reinforcement","title":"Autonomous Driving using Safe Reinforcement Learning by Incorporating a Regret-based Human Lane-Changing Decision Model","date":"2019-10-10","arxiv_id":"1910.04803","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-on-autonomous","title":"Safe Reinforcement Learning on Autonomous Vehicles","date":"2019-09-27","arxiv_id":"1910.00399","repositories_listed":0,"syntology":null},{"url":null,"slug":"posterior-variance-analysis-of-gaussian","title":"Posterior Variance Analysis of Gaussian Processes with Application to Average Learning Curves","date":"2019-06-04","arxiv_id":"1906.01404","repositories_listed":0,"syntology":null},{"url":null,"slug":"safer-deep-rl-with-shallow-mcts-a-case-study","title":"Safer Deep RL with Shallow MCTS: A Case Study in Pommerman","date":"2019-04-10","arxiv_id":"1904.05759","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-logic-guided-safe-reinforcement","title":"Temporal Logic Guided Safe Reinforcement Learning Using Control Barrier Functions","date":"2019-03-23","arxiv_id":"1903.09885","repositories_listed":0,"syntology":null},{"url":null,"slug":"parenting-safe-reinforcement-learning-from","title":"Parenting: Safe Reinforcement Learning from Human Input","date":"2019-02-18","arxiv_id":"1902.06766","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-physically-safe-reinforcement","title":"Towards Physically Safe Reinforcement Learning under Supervision","date":"2019-01-19","arxiv_id":"1901.06576","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-cross-entropy-method-for-safe","title":"Constrained Cross-Entropy Method for Safe Reinforcement Learning","date":"2018-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-model","title":"Safe Reinforcement Learning with Model Uncertainty Estimates","date":"2018-10-19","arxiv_id":"1810.08700","repositories_listed":0,"syntology":null}],"record_sha256":"324aacf0d073caf5595ddd82f779abd7d96bb7faff27d140f1455e03726afb44","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}