{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/71","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":71,"pages_in_order":152,"rows_per_page":100,"rows":[7001,7100],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/70","next":"/task/reinforcement-learning-1/papers/72","papers":[{"url":null,"slug":"policy-optimization-for-continuous","title":"Policy Optimization for Continuous Reinforcement Learning","date":"2023-05-30","arxiv_id":"2305.18901","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-model-based-control-using-on-demand","title":"RL + Model-based Control: Using On-demand Optimal Control to Learn Versatile Legged Locomotion","date":"2023-05-29","arxiv_id":"2305.17842","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlad-reinforcement-learning-from-pixels-for","title":"RLAD: Reinforcement Learning from Pixels for Autonomous Driving in Urban Environments","date":"2023-05-29","arxiv_id":"2305.18510","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-better-understanding-of","title":"Towards a Better Understanding of Representation Dynamics under TD-learning","date":"2023-05-29","arxiv_id":"2305.18491","repositories_listed":0,"syntology":null},{"url":null,"slug":"potential-based-credit-assignment-for","title":"Potential-based Credit Assignment for Cooperative RL-based Testing of Autonomous Vehicles","date":"2023-05-28","arxiv_id":"2305.18380","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-with-6","title":"Distributional Reinforcement Learning with Dual Expectile-Quantile Regression","date":"2023-05-26","arxiv_id":"2305.16877","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-agentic-transformer-from-chain-of","title":"Emergent Agentic Transformer from Chain of Hindsight Experience","date":"2023-05-26","arxiv_id":"2305.16554","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-interpretable-models-of-aircraft","title":"Learning Interpretable Models of Aircraft Handling Behaviour by Reinforcement Learning from Human Feedback","date":"2023-05-26","arxiv_id":"2305.16924","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-synthesis-and-reinforcement-learning","title":"Policy Synthesis and Reinforcement Learning for Discounted LTL","date":"2023-05-26","arxiv_id":"2305.17115","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-simple-sequence","title":"Reinforcement Learning with Simple Sequence Priors","date":"2023-05-26","arxiv_id":"2305.17109","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-curious-price-of-distributional","title":"The Curious Price of Distributional Robustness in Reinforcement Learning with a Generative Model","date":"2023-05-26","arxiv_id":"2305.16589","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-optimal-control-1","title":"Deterministic policy gradient based optimal control with probabilistic constraints","date":"2023-05-25","arxiv_id":"2305.15755","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mini-review-on-the-utilization-of","title":"A Mini Review on the utilization of Reinforcement Learning with OPC UA","date":"2023-05-24","arxiv_id":"2305.15113","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-invariant-set-enhanced-safe","title":"Control invariant set enhanced safe reinforcement learning: improved sampling efficiency, guaranteed stability and robustness","date":"2023-05-24","arxiv_id":"2305.15602","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-plasticity","title":"Deep Reinforcement Learning with Plasticity Injection","date":"2023-05-24","arxiv_id":"2305.15555","repositories_listed":0,"syntology":null},{"url":null,"slug":"matrix-estimation-for-offline-reinforcement","title":"Matrix Estimation for Offline Reinforcement Learning with Low-Rank Structure","date":"2023-05-24","arxiv_id":"2305.15621","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmented-random-search-for-multi-objective","title":"Combining Multi-Objective Bayesian Optimization with Reinforcement Learning for TinyML","date":"2023-05-23","arxiv_id":"2305.14109","repositories_listed":0,"syntology":null},{"url":null,"slug":"chemgymrl-an-interactive-framework-for","title":"ChemGymRL: An Interactive Framework for Reinforcement Learning for Digital Chemistry","date":"2023-05-23","arxiv_id":"2305.14177","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-proximal-policy-optimization","title":"Constrained Proximal Policy Optimization","date":"2023-05-23","arxiv_id":"2305.14216","repositories_listed":0,"syntology":null},{"url":null,"slug":"proximal-policy-gradient-arborescence-for","title":"Proximal Policy Gradient Arborescence for Quality Diversity Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.13795","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-action-supervision-in-reinforcement","title":"Adaptive action supervision in reinforcement learning from real-world multi-agent demonstrations","date":"2023-05-22","arxiv_id":"2305.13030","repositories_listed":0,"syntology":null},{"url":null,"slug":"hjb-based-online-safe-reinforcement-learning","title":"Lagrangian-based online safe reinforcement learning for state-constrained systems","date":"2023-05-22","arxiv_id":"2305.12967","repositories_listed":0,"syntology":null},{"url":null,"slug":"invictus-optimizing-boolean-logic-circuit","title":"INVICTUS: Optimizing Boolean Logic Circuit Synthesis via Synergistic Learning and Search","date":"2023-05-22","arxiv_id":"2305.13164","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-primal-dual-reinforcement-learning","title":"Offline Primal-Dual Reinforcement Learning for Linear MDPs","date":"2023-05-22","arxiv_id":"2305.12944","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-optimal-energy-management-strategy","title":"Towards Optimal Energy Management Strategy for Hybrid Electric Vehicle with Reinforcement Learning","date":"2023-05-21","arxiv_id":"2305.12365","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-adaptation-for-sample-efficient","title":"Model-based adaptation for sample efficient transfer in reinforcement learning control of parameter-varying systems","date":"2023-05-20","arxiv_id":"2305.12158","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-world-to-solve-social","title":"Understanding the World to Solve Social Dilemmas Using Multi-Agent Reinforcement Learning","date":"2023-05-19","arxiv_id":"2305.11358","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimistic-natural-policy-gradient-a-simple","title":"Optimistic Natural Policy Gradient: a Simple Efficient Policy Optimization Framework for Online RL","date":"2023-05-18","arxiv_id":"2305.11032","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-blessing-of-heterogeneity-in-federated-q","title":"The Blessing of Heterogeneity in Federated Q-Learning: Linear Speedup and Beyond","date":"2023-05-18","arxiv_id":"2305.10697","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-agnostic-fine-tuning-provable","title":"Reward-agnostic Fine-tuning: Provable Statistical Benefits of Hybrid Reinforcement Learning","date":"2023-05-17","arxiv_id":"2305.10282","repositories_listed":0,"syntology":null},{"url":null,"slug":"coagent-networks-generalized-and-scaled","title":"Coagent Networks: Generalized and Scaled","date":"2023-05-16","arxiv_id":"2305.09838","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperation-is-all-you-need","title":"Cooperation Is All You Need","date":"2023-05-16","arxiv_id":"2305.10449","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-safe-robot-control","title":"Reinforcement Learning for Safe Robot Control using Control Lyapunov Barrier Functions","date":"2023-05-16","arxiv_id":"2305.09793","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-theoretical-analysis-of-optimistic-proximal","title":"A Theoretical Analysis of Optimistic Proximal Policy Optimization in Linear Markov Decision Processes","date":"2023-05-15","arxiv_id":"2305.08841","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-qoe-aware-digital-twin","title":"Attention-based QoE-aware Digital Twin Empowered Edge Computing for Immersive Virtual Reality","date":"2023-05-15","arxiv_id":"2305.08569","repositories_listed":0,"syntology":null},{"url":null,"slug":"horizon-free-reinforcement-learning-in-1","title":"Horizon-free Reinforcement Learning in Adversarial Linear Mixture MDPs","date":"2023-05-15","arxiv_id":"2305.08359","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-communication-design-at-scale","title":"Task-Oriented Communication Design at Scale","date":"2023-05-15","arxiv_id":"2305.08481","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniform-pac-guarantees-for-model-based-rl","title":"Uniform-PAC Guarantees for Model-Based RL with Bounded Eluder Dimension","date":"2023-05-15","arxiv_id":"2305.08350","repositories_listed":0,"syntology":null},{"url":null,"slug":"delay-adapted-policy-optimization-and","title":"Delay-Adapted Policy Optimization and Improved Regret for Adversarial MDP with Delayed Bandit Feedback","date":"2023-05-13","arxiv_id":"2305.07911","repositories_listed":0,"syntology":null},{"url":"/paper/towards-generalizable-reinforcement-learning","slug":"towards-generalizable-reinforcement-learning","title":"Towards Generalizable Reinforcement Learning for Trade Execution","date":"2023-05-12","arxiv_id":"2307.11685","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/towards-generalizable-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2307.11685","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11685"}},"official":null}},{"url":null,"slug":"on-practical-robust-reinforcement-learning","title":"On Practical Robust Reinforcement Learning: Practical Uncertainty Set and Double-Agent Algorithm","date":"2023-05-11","arxiv_id":"2305.06657","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-memory-mapping-using-deep","title":"Optimizing Memory Mapping Using Deep Reinforcement Learning","date":"2023-05-11","arxiv_id":"2305.07440","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-option-dependent-analysis-of-regret","title":"An Option-Dependent Analysis of Regret Minimization Algorithms in Finite-Horizon Semi-Markov Decision Processes","date":"2023-05-10","arxiv_id":"2305.06936","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovery-of-optimal-quantum-error-correcting","title":"Discovery of Optimal Quantum Error Correcting Codes via Reinforcement Learning","date":"2023-05-10","arxiv_id":"2305.06378","repositories_listed":0,"syntology":null},{"url":null,"slug":"supplementing-gradient-based-reinforcement","title":"Supplementing Gradient-Based Reinforcement Learning with Simple Evolutionary Ideas","date":"2023-05-10","arxiv_id":"2305.07571","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessment-of-reinforcement-learning","title":"Assessment of Reinforcement Learning Algorithms for Nuclear Power Plant Fuel Optimization","date":"2023-05-09","arxiv_id":"2305.05812","repositories_listed":0,"syntology":null},{"url":"/paper/learnable-behavior-control-breaking-atari","slug":"learnable-behavior-control-breaking-atari","title":"Learnable Behavior Control: Breaking Atari Human World Records via Sample-Efficient Behavior Selection","date":"2023-05-09","arxiv_id":"2305.05239","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlocator-reinforcement-learning-for-bug","title":"RLocator: Reinforcement Learning for Bug Localization","date":"2023-05-09","arxiv_id":"2305.05586","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-enhanced-agents-for-interactive","title":"Knowledge-enhanced Agents for Interactive Text Games","date":"2023-05-08","arxiv_id":"2305.05091","repositories_listed":0,"syntology":null},{"url":null,"slug":"truncating-trajectories-in-monte-carlo","title":"Truncating Trajectories in Monte Carlo Reinforcement Learning","date":"2023-05-07","arxiv_id":"2305.04361","repositories_listed":0,"syntology":null},{"url":null,"slug":"replicating-complex-dialogue-policy-of-humans","title":"Replicating Complex Dialogue Policy of Humans via Offline Imitation Learning with Supervised Regularization","date":"2023-05-06","arxiv_id":"2305.03987","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-use-reinforcement-learning-to","title":"How to Use Reinforcement Learning to Facilitate Future Electricity Market Design? Part 1: A Paradigmatic Theory","date":"2023-05-04","arxiv_id":"2305.02485","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-use-reinforcement-learning-to-1","title":"How to Use Reinforcement Learning to Facilitate Future Electricity Market Design? Part 2: Method and Applications","date":"2023-05-04","arxiv_id":"2305.06921","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-population-assisted-off-policy","title":"Rethinking Population-assisted Off-policy Reinforcement Learning","date":"2023-05-04","arxiv_id":"2305.02949","repositories_listed":0,"syntology":null},{"url":null,"slug":"gym-precice-reinforcement-learning","title":"Gym-preCICE: Reinforcement Learning Environments for Active Flow Control","date":"2023-05-03","arxiv_id":"2305.02033","repositories_listed":0,"syntology":null},{"url":null,"slug":"validation-of-massively-parallel-adaptive","title":"Validation of massively-parallel adaptive testing using dynamic control matching","date":"2023-05-02","arxiv_id":"2305.01334","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-learning-of-policy-with-unknown","title":"Joint Learning of Policy with Unknown Temporal Constraints for Safe Reinforcement Learning","date":"2023-04-30","arxiv_id":"2305.00576","repositories_listed":0,"syntology":null},{"url":null,"slug":"relbot-a-transfer-learning-approach-to","title":"A Transfer Learning Approach to Minimize Reinforcement Learning Risks in Energy Optimization for Smart Buildings","date":"2023-04-30","arxiv_id":"2305.00365","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-federated-reinforcement-learning-framework","title":"A Federated Reinforcement Learning Framework for Link Activation in Multi-link Wi-Fi Networks","date":"2023-04-28","arxiv_id":"2304.14720","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-step-distributional-reinforcement","title":"One-Step Distributional Reinforcement Learning","date":"2023-04-27","arxiv_id":"2304.14421","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-agents-run-relay-race-with-strangers","title":"Can Agents Run Relay Race with Strangers? Generalization of RL to Out-of-Distribution Trajectories","date":"2023-04-26","arxiv_id":"2304.13424","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-criteria-hardware-trojan-detection-a","title":"Multi-criteria Hardware Trojan Detection: A Reinforcement Learning Approach","date":"2023-04-26","arxiv_id":"2304.13232","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-partial","title":"Reinforcement Learning with Partial Parametric Model Knowledge","date":"2023-04-26","arxiv_id":"2304.13223","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-closer-look-at-reward-decomposition-for","title":"A Closer Look at Reward Decomposition for High-Level Robotic Explanations","date":"2023-04-25","arxiv_id":"2304.12958","repositories_listed":0,"syntology":null},{"url":null,"slug":"loss-and-reward-weighing-for-increased","title":"Loss- and Reward-Weighting for Efficient Distributed Reinforcement Learning","date":"2023-04-25","arxiv_id":"2304.12778","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-extraction-attacks-against","title":"Model Extraction Attacks Against Reinforcement Learning Based Controllers","date":"2023-04-25","arxiv_id":"2304.13090","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-benefits-of-general-coverage","title":"What can online reinforcement learning with function approximation benefit from general coverage conditions?","date":"2023-04-25","arxiv_id":"2304.12886","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-dynamic-program-decompositions-of-static","title":"On Dynamic Programming Decompositions of Static Risk Measures in Markov Decision Processes","date":"2023-04-24","arxiv_id":"2304.12477","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-resilience-to-environment-poisoning","title":"Policy Resilience to Environment Poisoning Attacks on Reinforcement Learning","date":"2023-04-24","arxiv_id":"2304.12151","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-knowledge","title":"Reinforcement Learning with Knowledge Representation and Reasoning: A Brief Survey","date":"2023-04-24","arxiv_id":"2304.12090","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cubic-regularized-policy-newton-algorithm","title":"A Cubic-regularized Policy Newton Algorithm for Reinforcement Learning","date":"2023-04-21","arxiv_id":"2304.10951","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-symbolic-subsymbolic-and-hybrid","title":"A Review of Symbolic, Subsymbolic and Hybrid Methods for Sequential Decision Making","date":"2023-04-20","arxiv_id":"2304.10590","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-policy-gradient-method-for-pomdps","title":"End-to-End Policy Gradient Method for POMDPs and Explainable Agents","date":"2023-04-19","arxiv_id":"2304.09769","repositories_listed":0,"syntology":null},{"url":null,"slug":"fastrlap-a-system-for-learning-high-speed","title":"FastRLAP: A System for Learning High-Speed Driving via Deep RL and Autonomous Practicing","date":"2023-04-19","arxiv_id":"2304.09831","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-and-adapting-agile-locomotion-skills","title":"Learning and Adapting Agile Locomotion Skills by Transferring Experience","date":"2023-04-19","arxiv_id":"2304.09834","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-multi-agent-reinforcement-6","title":"Cooperative Multi-Agent Reinforcement Learning for Inventory Management","date":"2023-04-18","arxiv_id":"2304.08769","repositories_listed":0,"syntology":null},{"url":null,"slug":"feasible-policy-iteration","title":"Feasible Policy Iteration for Safe Reinforcement Learning","date":"2023-04-18","arxiv_id":"2304.08845","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-feedback-efficient-reinforcement","title":"Provably Feedback-Efficient Reinforcement Learning via Active Reward Learning","date":"2023-04-18","arxiv_id":"2304.08944","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-self","title":"An adaptive safety layer with hard constraints for safe reinforcement learning in multi-energy management systems","date":"2023-04-18","arxiv_id":"2304.08897","repositories_listed":0,"syntology":null},{"url":null,"slug":"mddl-a-framework-for-reinforcement-learning","title":"MDDL: A Framework for Reinforcement Learning-based Position Allocation in Multi-Channel Feed","date":"2023-04-17","arxiv_id":"2304.09087","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-on-longitudinal-car-following-model","title":"Car-Following Models: A Multidisciplinary Review","date":"2023-04-14","arxiv_id":"2304.07143","repositories_listed":0,"syntology":null},{"url":null,"slug":"bandit-based-policy-invariant-explicit","title":"Bandit-Based Policy Invariant Explicit Shaping for Incorporating External Advice in Reinforcement Learning","date":"2023-04-14","arxiv_id":"2304.07163","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-optimal-reward-agnostic-exploration","title":"Minimax-Optimal Reward-Agnostic Exploration in Reinforcement Learning","date":"2023-04-14","arxiv_id":"2304.07278","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-decision-making-in-spatial-learning-a","title":"Exploring the Noise Resilience of Successor Features and Predecessor Features Algorithms in One and Two-Dimensional Environments","date":"2023-04-14","arxiv_id":"2304.06894","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-controllable-diffusion-models-via","title":"Towards Controllable Diffusion Models via Reward-Guided Exploration","date":"2023-04-14","arxiv_id":"2304.07132","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-dynamic-shielding-for-safe-and","title":"Model-based Dynamic Shielding for Safe and Efficient Multi-Agent Reinforcement Learning","date":"2023-04-13","arxiv_id":"2304.06281","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-intrinsic-stochasticity-of-real","title":"Facilitating Sim-to-real by Intrinsic Stochasticity of Real-Time Simulation in Reinforcement Learning for Robot Manipulation","date":"2023-04-12","arxiv_id":"2304.06056","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-robot-skill-transfer-with-enhanced","title":"Human-Robot Skill Transfer with Enhanced Compliance via Dynamic Movement Primitives","date":"2023-04-12","arxiv_id":"2304.05703","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-policy-reciprocity-with","title":"Multi-agent Policy Reciprocity with Theoretical Guarantee","date":"2023-04-12","arxiv_id":"2304.05632","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-invariant-set-enhanced-reinforcement","title":"Control invariant set enhanced reinforcement learning for process control: improved sampling efficiency and guaranteed stability","date":"2023-04-11","arxiv_id":"2304.05509","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-interpretability-performance-trade","title":"Optimal Interpretability-Performance Trade-off of Classification Trees with Black-Box Reinforcement Learning","date":"2023-04-11","arxiv_id":"2304.05839","repositories_listed":0,"syntology":null},{"url":null,"slug":"for-pre-trained-vision-models-in-motor","title":"For Pre-Trained Vision Models in Motor Control, Not All Policy Learning Methods are Created Equal","date":"2023-04-10","arxiv_id":"2304.04591","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-a-universal-human-prior-for","title":"Learning a Universal Human Prior for Dexterous Manipulation from Human Preference","date":"2023-04-10","arxiv_id":"2304.04602","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-driven-resource-allocation-in-optical","title":"AI-Driven Resource Allocation in Optical Wireless Communication Systems","date":"2023-04-08","arxiv_id":"2304.03880","repositories_listed":0,"syntology":null},{"url":null,"slug":"dream-adaptive-reinforcement-learning-based","title":"DREAM: Adaptive Reinforcement Learning based on Attention Mechanism for Temporal Knowledge Graph Reasoning","date":"2023-04-08","arxiv_id":"2304.03984","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolving-reinforcement-learning-environment","title":"Evolving Reinforcement Learning Environment to Minimize Learner's Achievable Reward: An Application on Hardening Active Directory Systems","date":"2023-04-08","arxiv_id":"2304.03998","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-input-embedding-size-search-for","title":"Continuous Input Embedding Size Search For Recommender Systems","date":"2023-04-07","arxiv_id":"2304.03501","repositories_listed":0,"syntology":null},{"url":null,"slug":"persuading-to-prepare-for-quitting-smoking","title":"Persuading to Prepare for Quitting Smoking with a Virtual Coach: Using States and User Characteristics to Predict Behavior","date":"2023-04-05","arxiv_id":"2304.02264","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multiagent-cyberbattlesim-for-rl-cyber","title":"A Multiagent CyberBattleSim for RL Cyber Operation Agents","date":"2023-04-03","arxiv_id":"2304.11052","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-tutorial-introduction-to-reinforcement","title":"A Tutorial Introduction to Reinforcement Learning","date":"2023-04-03","arxiv_id":"2304.00803","repositories_listed":0,"syntology":null}],"record_sha256":"a368079290e4e2d4d72ef5bc2ceb8266000718a6df769f6ef4873b4c1543065e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}