{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/76","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":76,"pages_in_order":152,"rows_per_page":100,"rows":[7501,7600],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/75","next":"/task/reinforcement-learning-1/papers/77","papers":[{"url":null,"slug":"representation-learning-in-deep-rl-via","title":"Representation Learning in Deep RL via Discrete Information Bottleneck","date":"2022-12-28","arxiv_id":"2212.13835","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-the-linear-programming-framework","title":"Offline Reinforcement Learning via Linear-Programming with Error-Bound Induced Constraints","date":"2022-12-28","arxiv_id":"2212.13861","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-automating-codenames-spymasters-with","title":"Towards automating Codenames spymasters with deep reinforcement learning","date":"2022-12-28","arxiv_id":"2212.14104","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-learning-abstractions-via","title":"Towards Learning Abstractions via Reinforcement Learning","date":"2022-12-28","arxiv_id":"2212.13980","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-control-of-covid-19-in-buildings","title":"Data-driven control of COVID-19 in buildings: a reinforcement-learning approach","date":"2022-12-27","arxiv_id":"2212.13559","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-with-2","title":"Model-Based Reinforcement Learning with Multinomial Logistic Function Approximation","date":"2022-12-27","arxiv_id":"2212.13540","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-scheduling-of-island-integrated","title":"Optimal scheduling of island integrated energy systems considering multi-uncertainties and hydrothermal simultaneous transmission: A deep reinforcement learning approach","date":"2022-12-27","arxiv_id":"2212.13472","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-reinforcement-learning-with-loss","title":"Off-Policy Reinforcement Learning with Loss Function Weighted by Temporal Difference Error","date":"2022-12-26","arxiv_id":"2212.13175","repositories_listed":0,"syntology":null},{"url":null,"slug":"novel-reinforcement-learning-algorithm-for","title":"Novel Reinforcement Learning Algorithm for Suppressing Synchronization in Closed Loop Deep Brain Stimulators","date":"2022-12-25","arxiv_id":"2212.13260","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-heat-pump","title":"Deep Reinforcement Learning for Heat Pump Control","date":"2022-12-24","arxiv_id":"2212.12716","repositories_listed":0,"syntology":null},{"url":null,"slug":"shiro-soft-hierarchical-reinforcement","title":"SHIRO: Soft Hierarchical Reinforcement Learning","date":"2022-12-24","arxiv_id":"2212.12786","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-traffic-flow-prediction-based-on","title":"Streaming Traffic Flow Prediction Based on Continuous Reinforcement Learning","date":"2022-12-24","arxiv_id":"2212.12767","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-complexity-gains-of-single","title":"Understanding the Complexity Gains of Single-Task RL with a Curriculum","date":"2022-12-24","arxiv_id":"2212.12809","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-reinforcement-learning-for","title":"Investigation of reinforcement learning for shape optimization of profile extrusion dies","date":"2022-12-23","arxiv_id":"2212.12207","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-for-human","title":"Offline Reinforcement Learning for Human-Guided Human-Machine Interaction with Private Information","date":"2022-12-23","arxiv_id":"2212.12167","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-learned-simulation-environment-to-model-1","title":"A Learned Simulation Environment to Model Student Engagement and Retention in Automated Online Courses","date":"2022-12-22","arxiv_id":"2212.14693","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-surface-codes-with-deep","title":"Decoding surface codes with deep reinforcement learning and probabilistic policy reuse","date":"2022-12-22","arxiv_id":"2212.11890","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-approaches-to","title":"Reinforcement Learning Based Approaches to Adaptive Context Caching in Distributed Context Management Systems","date":"2022-12-22","arxiv_id":"2212.11709","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-memetic-algorithm-with-reinforcement","title":"A Memetic Algorithm with Reinforcement Learning for Sociotechnical Production Scheduling","date":"2022-12-21","arxiv_id":"2212.10936","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-is-not-enough-robustifying","title":"Imitation Is Not Enough: Robustifying Imitation with Reinforcement Learning for Challenging Driving Scenarios","date":"2022-12-21","arxiv_id":"2212.11419","repositories_listed":0,"syntology":null},{"url":null,"slug":"neighboring-state-based-rl-exploration","title":"Neighboring state-based RL Exploration","date":"2022-12-21","arxiv_id":"2212.10712","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-path-selection-in-software-defined","title":"Robust Path Selection in Software-defined WANs using Deep Reinforcement Learning","date":"2022-12-21","arxiv_id":"2212.11155","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-the-exploration-rate-for-value-of","title":"Adapting the Exploration Rate for Value-of-Information-Based Reinforcement Learning","date":"2022-12-20","arxiv_id":"2212.11083","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversar-adversarial-search-and-rescue-via","title":"AdverSAR: Adversarial Search and Rescue via Multi-Agent Reinforcement Learning","date":"2022-12-20","arxiv_id":"2212.10064","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-ai-dungeon-master-s-guide-learning-to","title":"I Cast Detect Thoughts: Learning to Converse and Guide with Intents and Theory-of-Mind in Dungeons and Dragons","date":"2022-12-20","arxiv_id":"2212.10060","repositories_listed":0,"syntology":null},{"url":null,"slug":"bandit-approach-to-conflict-free-multi-agent","title":"Bandit approach to conflict-free multi-agent Q-learning in view of photonic implementation","date":"2022-12-20","arxiv_id":"2212.09926","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-quantum-soft-actor-critic-for","title":"Variational Quantum Soft Actor-Critic for Robotic Arm Control","date":"2022-12-20","arxiv_id":"2212.11681","repositories_listed":0,"syntology":null},{"url":null,"slug":"dexterous-manipulation-from-images-autonomous","title":"Dexterous Manipulation from Images: Autonomous Real-World RL via Substep Guidance","date":"2022-12-19","arxiv_id":"2212.09902","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-for-text","title":"Inverse Reinforcement Learning for Text Summarization","date":"2022-12-19","arxiv_id":"2212.09917","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-policy-identification-in-active","title":"Near-optimal Policy Identification in Active Reinforcement Learning","date":"2022-12-19","arxiv_id":"2212.09510","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-policy-gradient-algorithms","title":"Quantum policy gradient algorithms","date":"2022-12-19","arxiv_id":"2212.09328","repositories_listed":0,"syntology":null},{"url":null,"slug":"taming-lagrangian-chaos-with-multi-objective","title":"Taming Lagrangian Chaos with Multi-Objective Reinforcement Learning","date":"2022-12-19","arxiv_id":"2212.09612","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-coreference-resolution-based-on","title":"Neural Coreference Resolution based on Reinforcement Learning","date":"2022-12-18","arxiv_id":"2212.09028","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-reinforcement-learning-with","title":"Risk-Sensitive Reinforcement Learning with Exponential Criteria","date":"2022-12-18","arxiv_id":"2212.09010","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-cyber-resilience-of-networked","title":"Enhancing Cyber Resilience of Networked Microgrids using Vertical Federated Reinforcement Learning","date":"2022-12-17","arxiv_id":"2212.08973","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-variable-representation-for","title":"Latent Variable Representation for Reinforcement Learning","date":"2022-12-17","arxiv_id":"2212.08765","repositories_listed":0,"syntology":null},{"url":null,"slug":"level-k-meta-learning-for-pedestrian-aware","title":"Cognitive Level-$k$ Meta-Learning for Safe and Pedestrian-Aware Autonomous Driving","date":"2022-12-17","arxiv_id":"2212.08800","repositories_listed":0,"syntology":null},{"url":null,"slug":"managing-temporal-resolution-in-continuous","title":"Managing Temporal Resolution in Continuous Value Estimation: A Fundamental Trade-off","date":"2022-12-17","arxiv_id":"2212.08949","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-trained-image-encoder-for-generalizable","title":"Pre-Trained Image Encoder for Generalizable Visual Reinforcement Learning","date":"2022-12-17","arxiv_id":"2212.08860","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-patrolling-with-battery","title":"An Energy-aware and Fault-tolerant Deep Reinforcement Learning based approach for Multi-agent Patrolling Problems","date":"2022-12-16","arxiv_id":"2212.08230","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-robot-reinforcement-learning-with","title":"Offline Robot Reinforcement Learning with Uncertainty-Guided Human Expert Sampling","date":"2022-12-16","arxiv_id":"2212.08232","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-agile-active","title":"Reinforcement Learning for Agile Active Target Sensing with a UAV","date":"2022-12-16","arxiv_id":"2212.08214","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-evaluation-for-offline-learning-are-we","title":"Safe Evaluation For Offline Learning: Are We Ready To Deploy?","date":"2022-12-16","arxiv_id":"2212.08302","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-offline-and-online","title":"Bridging the Gap Between Offline and Online Reinforcement Learning Evaluation Methodologies","date":"2022-12-15","arxiv_id":"2212.08131","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-information-seeking-exploration-and","title":"Active Inference and Reinforcement Learning: A unified inference on continuous state and action spaces under partial observability","date":"2022-12-15","arxiv_id":"2212.07946","repositories_listed":0,"syntology":null},{"url":null,"slug":"driver-assistance-eco-driving-and","title":"Driver Assistance Eco-driving and Transmission Control with Deep Reinforcement Learning","date":"2022-12-15","arxiv_id":"2212.07594","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-behaviors-in-multi-agent-target","title":"Emergent Behaviors in Multi-Agent Target Acquisition","date":"2022-12-15","arxiv_id":"2212.07891","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-with-4","title":"Multi-Agent Reinforcement Learning with Shared Resources for Inventory Management","date":"2022-12-15","arxiv_id":"2212.07684","repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-policy-learning-for-powertrain","title":"Residual Policy Learning for Powertrain Control","date":"2022-12-15","arxiv_id":"2212.07611","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-hardware-specific-automatic","title":"Towards Hardware-Specific Automatic Compression of Neural Networks","date":"2022-12-15","arxiv_id":"2212.07818","repositories_listed":0,"syntology":null},{"url":null,"slug":"ungeneralizable-contextual-logistic-bandit-in","title":"Reinforcement Learning in Credit Scoring and Underwriting","date":"2022-12-15","arxiv_id":"2212.07632","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-domain-transfer-via-semantic-skill","title":"Cross-Domain Transfer via Semantic Skill Imitation","date":"2022-12-14","arxiv_id":"2212.07407","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-exploration-in-resource-restricted","title":"Efficient Exploration in Resource-Restricted Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.06988","repositories_listed":0,"syntology":null},{"url":null,"slug":"explaining-agent-s-decision-making-in-a","title":"Explaining Agent's Decision-making in a Hierarchical Reinforcement Learning Scenario","date":"2022-12-14","arxiv_id":"2212.06967","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-strategies-for-cooperative-multi","title":"Hierarchical Strategies for Cooperative Multi-Agent Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07397","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-control-based-on-deep-reinforcement","title":"Quantum Control based on Deep Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07385","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-system","title":"Reinforcement Learning in System Identification","date":"2022-12-14","arxiv_id":"2212.07123","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-correction-from-baseline-towards-the","title":"Safety Correction from Baseline: Towards the Risk-aware Policy in Robotics via Dual-agent Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.06998","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-marginalized-importance-sampling-to","title":"Scaling Marginalized Importance Sampling to High-Dimensional State-Spaces via State Abstraction","date":"2022-12-14","arxiv_id":"2212.07486","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-off-policy-evaluation-in","title":"A Review of Off-Policy Evaluation in Reinforcement Learning","date":"2022-12-13","arxiv_id":"2212.06355","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-generalization-in-reinforcement-1","title":"Improving generalization in reinforcement learning through forked agents","date":"2022-12-13","arxiv_id":"2212.06451","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-approach-to-fair-solar-pv","title":"Model-Free Approach to Fair Solar PV Curtailment Using Reinforcement Learning","date":"2022-12-13","arxiv_id":"2212.06542","repositories_listed":0,"syntology":null},{"url":null,"slug":"ppo-ue-proximal-policy-optimization-via","title":"PPO-UE: Proximal Policy Optimization via Uncertainty-Aware Exploration","date":"2022-12-13","arxiv_id":"2212.06343","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-and-sample-efficient-distributed","title":"Scalable and Sample Efficient Distributed Policy Gradient Algorithms in Multi-Agent Networked Systems","date":"2022-12-13","arxiv_id":"2212.06357","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-cell-training-on-architecture-search","title":"Single Cell Training on Architecture Search for Image Denoising","date":"2022-12-13","arxiv_id":"2212.06368","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-reinforcement-learning-security","title":"A Survey on Reinforcement Learning Security with Application to Autonomous Driving","date":"2022-12-12","arxiv_id":"2212.06123","repositories_listed":0,"syntology":null},{"url":null,"slug":"corruption-robust-algorithms-with-uncertainty","title":"Corruption-Robust Algorithms with Uncertainty Weighting for Nonlinear Contextual Bandits and Markov Decision Processes","date":"2022-12-12","arxiv_id":"2212.05949","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-model-free-reinforcement-learning","title":"Evaluating Model-free Reinforcement Learning toward Safety-critical Tasks","date":"2022-12-12","arxiv_id":"2212.05727","repositories_listed":0,"syntology":null},{"url":null,"slug":"nearly-minimax-optimal-reinforcement-learning-2","title":"Nearly Minimax Optimal Reinforcement Learning for Linear Markov Decision Processes","date":"2022-12-12","arxiv_id":"2212.06132","repositories_listed":0,"syntology":null},{"url":null,"slug":"variance-reduced-conservative-policy","title":"Variance-Reduced Conservative Policy Iteration","date":"2022-12-12","arxiv_id":"2212.06283","repositories_listed":0,"syntology":null},{"url":null,"slug":"vo-q-l-towards-optimal-regret-in-model-free","title":"VO$Q$L: Towards Optimal Regret in Model-free RL with Nonlinear Function Approximation","date":"2022-12-12","arxiv_id":"2212.06069","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalization-through-the-lens-of-learning","title":"Generalization Through the Lens of Learning Dynamics","date":"2022-12-11","arxiv_id":"2212.05377","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-deep-reinforcement-learning-for-1","title":"Hierarchical Deep Reinforcement Learning for VWAP Strategy Optimization","date":"2022-12-11","arxiv_id":"2212.14670","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-deep-reinforcement-learning-1","title":"Off-Policy Deep Reinforcement Learning Algorithms for Handling Various Robotic Manipulator Tasks","date":"2022-12-11","arxiv_id":"2212.05572","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-modality-specific-representations","title":"Leveraging Modality-specific Representations for Audio-visual Speech Recognition via Reinforcement Learning","date":"2022-12-10","arxiv_id":"2212.05301","repositories_listed":0,"syntology":null},{"url":null,"slug":"relate-to-predict-towards-task-independent","title":"Relate to Predict: Towards Task-Independent Knowledge Representations for Reinforcement Learning","date":"2022-12-10","arxiv_id":"2212.05298","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-differentially-private","title":"Near-Optimal Differentially Private Reinforcement Learning","date":"2022-12-09","arxiv_id":"2212.04680","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-and-mixed-integer","title":"Reinforcement Learning and Mixed-Integer Programming for Power Plant Scheduling in Low Carbon Systems: Comparison and Hybridisation","date":"2022-12-09","arxiv_id":"2212.04824","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-predicting-traffic","title":"Reinforcement Learning for Predicting Traffic Accidents","date":"2022-12-09","arxiv_id":"2212.04677","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-stochastic-gradient-descent-algorithm","title":"A Novel Stochastic Gradient Descent Algorithm for Learning Principal Subspaces","date":"2022-12-08","arxiv_id":"2212.04025","repositories_listed":0,"syntology":null},{"url":null,"slug":"confidence-conditioned-value-functions-for","title":"Confidence-Conditioned Value Functions for Offline Reinforcement Learning","date":"2022-12-08","arxiv_id":"2212.04607","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-and-planning-of-flexible-mobile-micro","title":"Design and Planning of Flexible Mobile Micro-Grids Using Deep Reinforcement Learning","date":"2022-12-08","arxiv_id":"2212.04136","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-method-for-reinforcement-learning","title":"Enhanced method for reinforcement learning based dynamic obstacle avoidance by assessment of collision risk","date":"2022-12-08","arxiv_id":"2212.04123","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-resilient-power","title":"Reinforcement Learning for Resilient Power Grids","date":"2022-12-08","arxiv_id":"2212.04069","repositories_listed":0,"syntology":null},{"url":null,"slug":"system-design-for-an-integrated-lifelong","title":"System Design for an Integrated Lifelong Reinforcement Learning Agent for Real-Time Strategy Games","date":"2022-12-08","arxiv_id":"2212.04603","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-self-imitation-learning-from","title":"Accelerating Self-Imitation Learning from Demonstrations via Policy Constraints and Q-Ensemble","date":"2022-12-07","arxiv_id":"2212.03562","repositories_listed":0,"syntology":null},{"url":null,"slug":"selector-enhancer-learning-dynamic-selection","title":"Selector-Enhancer: Learning Dynamic Selection of Local and Non-local Attention Operation for Speech Enhancement","date":"2022-12-07","arxiv_id":"2212.03408","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-learned-simulation-environment-to-model","title":"A Learned Simulation Environment to Model Plant Growth in Indoor Farming","date":"2022-12-06","arxiv_id":"2212.03155","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-deep-reinforcement-learning-based-1","title":"A Novel Deep Reinforcement Learning Based Automated Stock Trading System Using Cascaded LSTM Networks","date":"2022-12-06","arxiv_id":"2212.02721","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-classification-of-moving-targets-with","title":"Active Classification of Moving Targets with Learned Control Policies","date":"2022-12-06","arxiv_id":"2212.03068","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-learning-of-voltage-control","title":"Efficient Learning of Voltage Control Strategies via Model-based Deep Reinforcement Learning","date":"2022-12-06","arxiv_id":"2212.02715","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-preference-learning-for-human-in-the","title":"Few-Shot Preference Learning for Human-in-the-Loop RL","date":"2022-12-06","arxiv_id":"2212.03363","repositories_listed":0,"syntology":null},{"url":null,"slug":"first-go-then-post-explore-the-benefits-of","title":"First Go, then Post-Explore: the Benefits of Post-Exploration in Intrinsic Motivation","date":"2022-12-06","arxiv_id":"2212.03251","repositories_listed":0,"syntology":null},{"url":null,"slug":"misspecification-in-inverse-reinforcement","title":"Misspecification in Inverse Reinforcement Learning","date":"2022-12-06","arxiv_id":"2212.03201","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-uav-control-with","title":"Reinforcement Learning for UAV control with Policy and Reward Shaping","date":"2022-12-06","arxiv_id":"2212.03828","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-inverse-reinforcement-learning-via","title":"Safe Inverse Reinforcement Learning via Control Barrier Function","date":"2022-12-06","arxiv_id":"2212.02753","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-planning-and-learning-framework","title":"Scalable Planning and Learning Framework Development for Swarm-to-Swarm Engagement Problems","date":"2022-12-06","arxiv_id":"2212.02909","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-self-predictive-learning-for","title":"Understanding Self-Predictive Learning for Reinforcement Learning","date":"2022-12-06","arxiv_id":"2212.03319","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-deep-reinforcement-learning","title":"A Hierarchical Deep Reinforcement Learning Framework for 6-DOF UCAV Air-to-Air Combat","date":"2022-12-05","arxiv_id":"2212.03830","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-look-at-risk","title":"Robust Reinforcement Learning for Risk-Sensitive Linear Quadratic Gaussian Control","date":"2022-12-05","arxiv_id":"2212.02072","repositories_listed":0,"syntology":null}],"record_sha256":"7ef2a860863437cb686322a6e717901556b323e779e6c34da823c9a481cbd638","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}