{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/64","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":64,"pages_in_order":135,"rows_per_page":100,"rows":[6301,6400],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/63","next":"/task/reinforcement-learning-2/papers/65","papers":[{"url":null,"slug":"reinforced-self-training-rest-for-language","title":"Reinforced Self-Training (ReST) for Language Modeling","date":"2023-08-17","arxiv_id":"2308.08998","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-battery-management","title":"Reinforcement Learning for Battery Management in Dairy Farming","date":"2023-08-17","arxiv_id":"2308.09023","repositories_listed":0,"syntology":null},{"url":null,"slug":"reprohrl-towards-multi-goal-navigation-in-the","title":"ReProHRL: Towards Multi-Goal Navigation in the Real World using Hierarchical Agents","date":"2023-08-17","arxiv_id":"2308.08737","repositories_listed":0,"syntology":null},{"url":null,"slug":"eliciting-risk-aversion-with-inverse","title":"Eliciting Risk Aversion with Inverse Reinforcement Learning via Interactive Questioning","date":"2023-08-16","arxiv_id":"2308.08427","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-renewable-energy-in-agriculture-a","title":"Integrating Renewable Energy in Agriculture: A Deep Reinforcement Learning-based Approach","date":"2023-08-16","arxiv_id":"2308.08611","repositories_listed":0,"syntology":null},{"url":null,"slug":"partially-observable-multi-agent-rl-with","title":"Partially Observable Multi-Agent Reinforcement Learning with Information Sharing","date":"2023-08-16","arxiv_id":"2308.08705","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-process-2","title":"Deep reinforcement learning for process design: Review and perspective","date":"2023-08-15","arxiv_id":"2308.07822","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-knowledge-from-resource-management","title":"Distilling Knowledge from Resource Management Algorithms to Neural Networks: A Unified Training Assistance Approach","date":"2023-08-15","arxiv_id":"2308.07511","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-personas-for-games-with-multimodal","title":"Generating Personas for Games with Multimodal Adversarial Imitation Learning","date":"2023-08-15","arxiv_id":"2308.07598","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-robot-challenge-2022-learning-dexterous","title":"Real Robot Challenge 2022: Learning Dexterous Manipulation from Offline Data in the Real World","date":"2023-08-15","arxiv_id":"2308.07741","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-rl-augmented-cold","title":"On-demand Cold Start Frequency Reduction with Off-Policy Reinforcement Learning in Serverless Computing","date":"2023-08-15","arxiv_id":"2308.07541","repositories_listed":0,"syntology":null},{"url":null,"slug":"insurance-pricing-on-price-comparison","title":"Insurance pricing on price comparison websites via reinforcement learning","date":"2023-08-14","arxiv_id":"2308.06935","repositories_listed":0,"syntology":null},{"url":null,"slug":"intune-reinforcement-learning-based-data","title":"InTune: Reinforcement Learning-based Data Pipeline Optimization for Deep Recommendation Models","date":"2023-08-13","arxiv_id":"2308.08500","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-solution-and-concrete-implementation","title":"A new solution and concrete implementation steps for Artificial General Intelligence","date":"2023-08-12","arxiv_id":"2308.09721","repositories_listed":0,"syntology":null},{"url":null,"slug":"cyberforce-a-federated-reinforcement-learning","title":"CyberForce: A Federated Reinforcement Learning Framework for Malware Mitigation","date":"2023-08-11","arxiv_id":"2308.05978","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-control-policies-for-variable","title":"Learning Control Policies for Variable Objectives from Offline Data","date":"2023-08-11","arxiv_id":"2308.06127","repositories_listed":0,"syntology":null},{"url":null,"slug":"safeguarding-learning-based-control-for-smart","title":"Safeguarding Learning-based Control for Smart Energy Systems with Sampling Specifications","date":"2023-08-11","arxiv_id":"2308.06069","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-classical-and-deep","title":"A Comparison of Classical and Deep Reinforcement Learning Methods for HVAC Control","date":"2023-08-10","arxiv_id":"2308.05711","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-deep-reinforcement-learning-for-1","title":"Adversarial Deep Reinforcement Learning for Cyber Security in Software Defined Networks","date":"2023-08-09","arxiv_id":"2308.04909","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-autonomous-separation-assurance","title":"Improving Autonomous Separation Assurance through Distributed Reinforcement Learning with Attention Networks","date":"2023-08-09","arxiv_id":"2308.04958","repositories_listed":0,"syntology":null},{"url":null,"slug":"characterization-of-human-balance-through-a","title":"Characterization of Human Balance through a Reinforcement Learning-based Muscle Controller","date":"2023-08-08","arxiv_id":"2308.04462","repositories_listed":0,"syntology":null},{"url":null,"slug":"heterogeneous-360-degree-videos-in-metaverse","title":"Heterogeneous 360 Degree Videos in Metaverse: Differentiated Reinforcement Learning Approaches","date":"2023-08-08","arxiv_id":"2308.04083","repositories_listed":0,"syntology":null},{"url":null,"slug":"scope-loss-for-imbalanced-classification-and","title":"Scope Loss for Imbalanced Classification and RL Exploration","date":"2023-08-08","arxiv_id":"2308.04024","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-network-for-stochastic-process","title":"Deep Q-Network for Stochastic Process Environments","date":"2023-08-07","arxiv_id":"2308.03316","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-generalization-in-offline","title":"Exploiting Generalization in Offline Reinforcement Learning via Unseen State Augmentations","date":"2023-08-07","arxiv_id":"2308.03882","repositories_listed":0,"syntology":null},{"url":null,"slug":"surrogate-empowered-sim2real-transfer-of-deep","title":"Surrogate Empowered Sim2Real Transfer of Deep Reinforcement Learning for ORC Superheat Control","date":"2023-08-05","arxiv_id":"2308.02765","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-optimal-admission-control-in","title":"Learning Optimal Admission Control in Partially Observable Queueing Networks","date":"2023-08-04","arxiv_id":"2308.02391","repositories_listed":0,"syntology":null},{"url":null,"slug":"nonprehensile-planar-manipulation-through","title":"Nonprehensile Planar Manipulation through Reinforcement Learning with Multimodal Categorical Exploration","date":"2023-08-04","arxiv_id":"2308.02459","repositories_listed":0,"syntology":null},{"url":null,"slug":"vehicles-control-collision-avoidance-using","title":"Vehicles Control: Collision Avoidance using Federated Deep Reinforcement Learning","date":"2023-08-04","arxiv_id":"2308.02614","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligning-agent-policy-with-externalities","title":"PARL: A Unified Framework for Policy Alignment in Reinforcement Learning from Human Feedback","date":"2023-08-03","arxiv_id":"2308.02585","repositories_listed":0,"syntology":null},{"url":null,"slug":"avoidance-navigation-based-on-offline-pre","title":"Avoidance Navigation Based on Offline Pre-Training Reinforcement Learning","date":"2023-08-03","arxiv_id":"2308.01551","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-reinforcement-learning-of-koopman","title":"End-to-End Reinforcement Learning of Koopman Models for Economic Nonlinear Model Predictive Control","date":"2023-08-03","arxiv_id":"2308.01674","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-reinforcement-learning-for","title":"Investigating Reinforcement Learning for Communication Strategies in a Task-Initiative Setting","date":"2023-08-03","arxiv_id":"2308.01479","repositories_listed":0,"syntology":null},{"url":null,"slug":"marlim-multi-agent-reinforcement-learning-for","title":"MARLIM: Multi-Agent Reinforcement Learning for Inventory Management","date":"2023-08-03","arxiv_id":"2308.01649","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-multi-agent-reinforcement-learning-1","title":"Quantum Multi-Agent Reinforcement Learning for Autonomous Mobility Cooperation","date":"2023-08-03","arxiv_id":"2308.01519","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-the-solo12-quadruped-robot-with","title":"Controlling the Solo12 Quadruped Robot with Deep Reinforcement Learning","date":"2023-08-02","arxiv_id":"2309.16683","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-gradient-temporal-difference-learning","title":"Revisiting a Design Choice in Gradient Temporal Difference Learning","date":"2023-08-02","arxiv_id":"2308.01170","repositories_listed":0,"syntology":null},{"url":null,"slug":"follow-the-soldiers-with-optimized-single","title":"Follow the Soldiers with Optimized Single-Shot Multibox Detection and Reinforcement Learning","date":"2023-08-02","arxiv_id":"2308.01389","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-generalization-in-visual","title":"Improving Generalization in Visual Reinforcement Learning via Conflict-aware Gradient Agreement Augmentation","date":"2023-08-02","arxiv_id":"2308.01194","repositories_listed":0,"syntology":null},{"url":null,"slug":"wasserstein-diversity-enriched-regularizer","title":"Wasserstein Diversity-Enriched Regularizer for Hierarchical Reinforcement Learning","date":"2023-08-02","arxiv_id":"2308.00989","repositories_listed":0,"syntology":null},{"url":null,"slug":"pixel-to-policy-dqn-encoders-for-within-cross","title":"Pixel to policy: DQN Encoders for within & cross-game reinforcement learning","date":"2023-08-01","arxiv_id":"2308.00318","repositories_listed":0,"syntology":null},{"url":null,"slug":"target-search-and-navigation-in-heterogeneous","title":"Target Search and Navigation in Heterogeneous Robot Systems with Deep Reinforcement Learning","date":"2023-08-01","arxiv_id":"2308.00331","repositories_listed":0,"syntology":null},{"url":null,"slug":"modulation-enhanced-excitation-for-continuous","title":"Modulation-Enhanced Excitation for Continuous-Time Reinforcement Learning via Symmetric Kronecker Products","date":"2023-07-31","arxiv_id":"2307.16862","repositories_listed":0,"syntology":null},{"url":null,"slug":"esp-exploiting-symmetry-prior-for-multi-agent","title":"ESP: Exploiting Symmetry Prior for Multi-Agent Reinforcement Learning","date":"2023-07-30","arxiv_id":"2307.16186","repositories_listed":0,"syntology":null},{"url":null,"slug":"rating-based-reinforcement-learning","title":"Rating-based Reinforcement Learning","date":"2023-07-30","arxiv_id":"2307.16348","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-electric-vehicle-balancing-of","title":"Robust Electric Vehicle Balancing of Autonomous Mobility-On-Demand System: A Multi-Agent Reinforcement Learning Approach","date":"2023-07-30","arxiv_id":"2307.16228","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-deep-reinforcement-learning-algorithm","title":"Dynamic deep-reinforcement-learning algorithm in Partially Observed Markov Decision Processes","date":"2023-07-29","arxiv_id":"2307.15931","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-under-probabilistic","title":"Reinforcement Learning Under Probabilistic Spatio-Temporal Constraints with Time Windows","date":"2023-07-29","arxiv_id":"2307.15910","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialogue-shaping-empowering-agents-through","title":"Dialogue Shaping: Empowering Agents through NPC Interaction","date":"2023-07-28","arxiv_id":"2307.15833","repositories_listed":0,"syntology":null},{"url":null,"slug":"primitive-skill-based-robot-learning-from","title":"Primitive Skill-based Robot Learning from Human Evaluative Feedback","date":"2023-07-28","arxiv_id":"2307.15801","repositories_listed":0,"syntology":null},{"url":null,"slug":"trackagent-6d-object-tracking-via","title":"TrackAgent: 6D Object Tracking via Reinforcement Learning","date":"2023-07-28","arxiv_id":"2307.15671","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-ensemble-method-of-deep-reinforcement","title":"An Ensemble Method of Deep Reinforcement Learning for Automated Cryptocurrency Trading","date":"2023-07-27","arxiv_id":"2309.00626","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-safety-constraints-in","title":"Evaluation of Safety Constraints in Autonomous Navigation with Deep Reinforcement Learning","date":"2023-07-27","arxiv_id":"2307.14568","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-problems-and-fundamental-limitations-of","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","date":"2023-07-27","arxiv_id":"2307.15217","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-guided-fuzz-testing","title":"Reinforcement learning guided fuzz testing for a browser's HTML rendering engine","date":"2023-07-27","arxiv_id":"2307.14556","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-international-climate-policy-via-1","title":"Improving International Climate Policy via Mutually Conditional Binding Commitments","date":"2023-07-26","arxiv_id":"2307.14266","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-offline-reinforcement-learning","title":"Integrating Offline Reinforcement Learning with Transformers for Sequential Recommendation","date":"2023-07-26","arxiv_id":"2307.14450","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-by-guided-safe","title":"Reinforcement Learning by Guided Safe Exploration","date":"2023-07-26","arxiv_id":"2307.14316","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-efficient-orchestrations-for","title":"Communication-Efficient Orchestrations for URLLC Service via Hierarchical Reinforcement Learning","date":"2023-07-25","arxiv_id":"2307.13415","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-robust-goal","title":"Deep Reinforcement Learning for Robust Goal-Based Wealth Management","date":"2023-07-25","arxiv_id":"2307.13501","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-on-policy","title":"Offline Reinforcement Learning with On-Policy Q-Function Regularization","date":"2023-07-25","arxiv_id":"2307.13824","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-margins-for-reinforcement-learning","title":"Safety Margins for Reinforcement Learning","date":"2023-07-25","arxiv_id":"2307.13642","repositories_listed":0,"syntology":null},{"url":null,"slug":"settling-the-sample-complexity-of-online","title":"Settling the Sample Complexity of Online Reinforcement Learning","date":"2023-07-25","arxiv_id":"2307.13586","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-punctuation-restoration-with-data","title":"Boosting Punctuation Restoration with Data Generation and Reinforcement Learning","date":"2023-07-24","arxiv_id":"2307.12949","repositories_listed":0,"syntology":null},{"url":"/paper/parallel-q-learning-scaling-off-policy","slug":"parallel-q-learning-scaling-off-policy","title":"Parallel $Q$-Learning: Scaling Off-policy Reinforcement Learning under Massively Parallel Simulation","date":"2023-07-24","arxiv_id":"2307.12983","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parallel-q-learning-scaling-off-policy#ran","syntology_url":"https://syntology.ai/paper/2307.12983","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12983"}},"official":null}},{"url":null,"slug":"pilot-performance-modeling-via-observer-based","title":"Pilot Performance modeling via observer-based inverse reinforcement learning","date":"2023-07-24","arxiv_id":"2307.13150","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-agents-for-attacking-inaudible","title":"Adversarial Agents For Attacking Inaudible Voice Activated Devices","date":"2023-07-23","arxiv_id":"2307.12204","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-control-of-flow-over-rotating-cylinder","title":"Active Control of Flow over Rotating Cylinder by Multiple Jets using Deep Reinforcement Learning","date":"2023-07-22","arxiv_id":"2307.12083","repositories_listed":0,"syntology":null},{"url":null,"slug":"dip-rl-demonstration-inferred-preference","title":"DIP-RL: Demonstration-Inferred Preference Learning in Minecraft","date":"2023-07-22","arxiv_id":"2307.12158","repositories_listed":0,"syntology":null},{"url":null,"slug":"game-theoretic-robust-reinforcement-learning","title":"Game-Theoretic Robust Reinforcement Learning Handles Temporally-Coupled Perturbations","date":"2023-07-22","arxiv_id":"2307.12062","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-robot-bayesian-reinforcement-learning-for","title":"On-Robot Bayesian Reinforcement Learning for POMDPs","date":"2023-07-22","arxiv_id":"2307.11954","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlocking-carbon-reduction-potential-with","title":"Using Reinforcement Learning for the Three-Dimensional Loading Capacitated Vehicle Routing Problem","date":"2023-07-22","arxiv_id":"2307.12136","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-multi-agent-reinforcement","title":"An Analysis of Multi-Agent Reinforcement Learning for Decentralized Inventory Control Systems","date":"2023-07-21","arxiv_id":"2307.11432","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-system-for","title":"Deep Reinforcement Learning Based System for Intraoperative Hyperspectral Video Autofocusing","date":"2023-07-21","arxiv_id":"2307.11638","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-practical-reinforcement-learning-for","title":"Towards practical reinforcement learning for tokamak magnetic control","date":"2023-07-21","arxiv_id":"2307.11546","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-definition-of-continual-reinforcement","title":"A Definition of Continual Reinforcement Learning","date":"2023-07-20","arxiv_id":"2307.11046","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-adaptive-dual-level-reinforcement-learning","title":"An Adaptive Dual-level Reinforcement Learning Approach for Optimal Trade Execution","date":"2023-07-20","arxiv_id":"2307.10649","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-conditioned-reinforcement-learning-with-1","title":"Goal-Conditioned Reinforcement Learning with Disentanglement-based Reachability Planning","date":"2023-07-20","arxiv_id":"2307.10846","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-convergence-of-bounded-agents","title":"On the Convergence of Bounded Agents","date":"2023-07-20","arxiv_id":"2307.11044","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-credit-index","title":"Reinforcement Learning for Credit Index Option Hedging","date":"2023-07-19","arxiv_id":"2307.09844","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-driving-policy-learning-with-guided","title":"Robust Driving Policy Learning with Guided Meta Reinforcement Learning","date":"2023-07-19","arxiv_id":"2307.10160","repositories_listed":0,"syntology":null},{"url":null,"slug":"technical-challenges-of-deploying","title":"Technical Challenges of Deploying Reinforcement Learning Agents for Game Testing in AAA Games","date":"2023-07-19","arxiv_id":"2307.11105","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-time-reinforcement-learning-new","title":"Continuous-Time Reinforcement Learning: New Design Algorithms with Theoretical Insights and Performance Guarantees","date":"2023-07-18","arxiv_id":"2307.08920","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-of-small-spacecraft-by-optimal-output","title":"Control of Small Spacecraft by Optimal Output Regulation: A Reinforcement Learning Approach","date":"2023-07-18","arxiv_id":"2307.09428","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-cross-segmentation-for-improved","title":"Data Cross-Segmentation for Improved Generalization in Reinforcement Learning Based Algorithmic Trading","date":"2023-07-18","arxiv_id":"2307.09377","repositories_listed":0,"syntology":null},{"url":null,"slug":"qmnet-importance-aware-message-exchange-for","title":"QMNet: Importance-Aware Message Exchange for Decentralized Multi-Agent Reinforcement Learning","date":"2023-07-18","arxiv_id":"2307.09051","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multiobjective-reinforcement-learning","title":"A Multiobjective Reinforcement Learning Framework for Microgrid Energy Management","date":"2023-07-17","arxiv_id":"2307.08692","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-euclidean-symmetry-be-leveraged-in","title":"Can Euclidean Symmetry be Leveraged in Reinforcement Learning and Planning?","date":"2023-07-17","arxiv_id":"2307.08226","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-stationary-policy-learning-for-multi","title":"Non-Stationary Policy Learning for Multi-Timescale Multi-Agent Reinforcement Learning","date":"2023-07-17","arxiv_id":"2307.08794","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-accelerating-benders-decomposition","title":"Accelerating Cutting-Plane Algorithms via Reinforcement Learning Surrogates","date":"2023-07-17","arxiv_id":"2307.08816","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-user-types-mapping-user-traits-by","title":"Discovering User Types: Mapping User Traits by Task-Specific Behaviors in Reinforcement Learning","date":"2023-07-16","arxiv_id":"2307.08169","repositories_listed":0,"syntology":null},{"url":null,"slug":"magnetic-field-based-reward-shaping-for-goal","title":"Magnetic Field-Based Reward Shaping for Goal-Conditioned Reinforcement Learning","date":"2023-07-16","arxiv_id":"2307.08033","repositories_listed":0,"syntology":null},{"url":null,"slug":"aioptimizer-a-reinforcement-learning-based","title":"AIOptimizer - Software performance optimisation prototype for cost minimisation","date":"2023-07-15","arxiv_id":"2307.07846","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-model-predictive-control-and","title":"Combining model-predictive control and predictive reinforcement learning for stable quadrupedal robot locomotion","date":"2023-07-15","arxiv_id":"2307.07752","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-action-robust-reinforcement","title":"Efficient Action Robust Reinforcement Learning with Probabilistic Policy Execution Uncertainty","date":"2023-07-15","arxiv_id":"2307.07666","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-adversarial-attacks-on-online-multi","title":"Efficient Adversarial Attacks on Online Multi-agent Reinforcement Learning","date":"2023-07-15","arxiv_id":"2307.07670","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-is-not-believing-robust-reinforcement","title":"Seeing is not Believing: Robust Reinforcement Learning against Spurious Correlation","date":"2023-07-15","arxiv_id":"2307.07907","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-control-policy-for-artificial-pancreas","title":"Hybrid Control Policy for Artificial Pancreas via Ensemble Deep Reinforcement Learning","date":"2023-07-13","arxiv_id":"2307.06501","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-transformation-sequence-retrieval-with","title":"Image Transformation Sequence Retrieval with General Reinforcement Learning","date":"2023-07-13","arxiv_id":"2307.06630","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effective-horizon-of-inverse","title":"On the Effective Horizon of Inverse Reinforcement Learning","date":"2023-07-13","arxiv_id":"2307.06541","repositories_listed":0,"syntology":null}],"record_sha256":"e92af8c9fd6ffc30cc544b6a3a2f1c2f947445ada4f52618de67c54b45ba927f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}