{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/69","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":69,"pages_in_order":152,"rows_per_page":100,"rows":[6801,6900],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/68","next":"/task/reinforcement-learning-1/papers/70","papers":[{"url":null,"slug":"marginalized-importance-sampling-for-off","title":"Marginalized Importance Sampling for Off-Environment Policy Evaluation","date":"2023-09-04","arxiv_id":"2309.01807","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-aware-safety-for-interactive","title":"Deception Game: Closing the Safety-Learning Loop in Interactive Robot Autonomy","date":"2023-09-03","arxiv_id":"2309.01267","repositories_listed":0,"syntology":null},{"url":null,"slug":"neurosymbolic-reinforcement-learning-and","title":"Neurosymbolic Reinforcement Learning and Planning: A Survey","date":"2023-09-02","arxiv_id":"2309.01038","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-dexterity-chaining-dexterous","title":"Sequential Dexterity: Chaining Dexterous Policies for Long-Horizon Manipulation","date":"2023-09-02","arxiv_id":"2309.00987","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-lidar-driven-reinforcement","title":"End-to-end Lidar-Driven Reinforcement Learning for Autonomous Racing","date":"2023-09-01","arxiv_id":"2309.00296","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-decision-transformers-for","title":"Multi-Objective Decision Transformers for Offline Reinforcement Learning","date":"2023-08-31","arxiv_id":"2308.16379","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-construction","title":"A reinforcement learning based construction material supply strategy using robotic crane and computer vision for building reconstruction after an earthquake","date":"2023-08-30","arxiv_id":"2308.16280","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-robustness-and-generalization-in","title":"Benchmarking Robustness and Generalization in Multi-Agent Systems: A Case Study on Neural MMO","date":"2023-08-30","arxiv_id":"2308.15802","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-improvement-of-model-predictive","title":"On the improvement of model-predictive controllers","date":"2023-08-29","arxiv_id":"2308.15157","repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-progress-in-energy-management-of","title":"Recent Progress in Energy Management of Connected Hybrid Electric Vehicles Using Reinforcement Learning","date":"2023-08-28","arxiv_id":"2308.14602","repositories_listed":0,"syntology":null},{"url":null,"slug":"target-independent-xla-optimization-using","title":"Target-independent XLA optimization using Reinforcement Learning","date":"2023-08-28","arxiv_id":"2308.14364","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-optimal-control-2","title":"Reinforcement Learning-based Optimal Control and Software Rejuvenation for Safe and Efficient UAV Navigation","date":"2023-08-27","arxiv_id":"2308.14139","repositories_listed":0,"syntology":null},{"url":null,"slug":"jax-lob-a-gpu-accelerated-limit-order-book","title":"JAX-LOB: A GPU-Accelerated limit order book simulator to unlock large scale reinforcement learning for trading","date":"2023-08-25","arxiv_id":"2308.13289","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-band-assignment-and-beam-management","title":"Joint Band Assignment and Beam Management using Hierarchical Reinforcement Learning for Multi-Band Communication","date":"2023-08-25","arxiv_id":"2308.13202","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-assisted-evolutionary","title":"Reinforcement Learning-assisted Evolutionary Algorithm: A Survey and Research Opportunities","date":"2023-08-25","arxiv_id":"2308.13420","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-exploration-networks","title":"Bayesian Exploration Networks","date":"2023-08-24","arxiv_id":"2308.13049","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditional-kernel-imitation-learning-for","title":"Conditional Kernel Imitation Learning for Continuous State Environments","date":"2023-08-24","arxiv_id":"2308.12573","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-reinforcement-learning-based","title":"Continuous Reinforcement Learning-based Dynamic Difficulty Adjustment in a Visual Working Memory Game","date":"2023-08-24","arxiv_id":"2308.12726","repositories_listed":0,"syntology":null},{"url":null,"slug":"extreme-risk-mitigation-in-reinforcement","title":"Extreme Risk Mitigation in Reinforcement Learning using Extreme Value Theory","date":"2023-08-24","arxiv_id":"2308.13011","repositories_listed":0,"syntology":null},{"url":null,"slug":"racing-towards-reinforcement-learning-based","title":"Racing Towards Reinforcement Learning based control of an Autonomous Formula SAE Car","date":"2023-08-24","arxiv_id":"2308.13088","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-informed-evolutionary","title":"Reinforcement learning informed evolutionary search for autonomous systems testing","date":"2023-08-24","arxiv_id":"2308.12762","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-validating-long-term-user-feedbacks","title":"Towards Validating Long-Term User Feedbacks in Interactive Recommendation Systems","date":"2023-08-22","arxiv_id":"2308.11137","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-homogenization-approach-for-gradient","title":"A Homogenization Approach for Gradient-Dominated Stochastic Optimization","date":"2023-08-21","arxiv_id":"2308.10630","repositories_listed":0,"syntology":null},{"url":null,"slug":"soft-decomposed-policy-critic-bridging-the","title":"Soft Decomposed Policy-Critic: Bridging the Gap for Effective Continuous Control with Discrete RL","date":"2023-08-20","arxiv_id":"2308.10203","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-exact-combinatorial-optimization","title":"Accelerating Exact Combinatorial Optimization via RL-based Initialization -- A Case Study in Scheduling","date":"2023-08-19","arxiv_id":"2308.11652","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-assisted-semantic-communication-with","title":"UAV-assisted Semantic Communication with Hybrid Action Reinforcement Learning","date":"2023-08-18","arxiv_id":"2309.16713","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-integrated-sensing-and","title":"Data-driven Integrated Sensing and Communication: Recent Advances, Challenges, and Future Prospects","date":"2023-08-17","arxiv_id":"2308.09090","repositories_listed":0,"syntology":null},{"url":null,"slug":"imm-an-imitative-reinforcement-learning","title":"IMM: An Imitative Reinforcement Learning Approach with Predictive Representation Learning for Automatic Market Making","date":"2023-08-17","arxiv_id":"2308.08918","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-algorithm-with-improved-sample","title":"Improving Sample Efficiency of Model-Free Algorithms for Zero-Sum Markov Games","date":"2023-08-17","arxiv_id":"2308.08858","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-self-training-rest-for-language","title":"Reinforced Self-Training (ReST) for Language Modeling","date":"2023-08-17","arxiv_id":"2308.08998","repositories_listed":0,"syntology":null},{"url":null,"slug":"reprohrl-towards-multi-goal-navigation-in-the","title":"ReProHRL: Towards Multi-Goal Navigation in the Real World using Hierarchical Agents","date":"2023-08-17","arxiv_id":"2308.08737","repositories_listed":0,"syntology":null},{"url":null,"slug":"partially-observable-multi-agent-rl-with","title":"Partially Observable Multi-Agent Reinforcement Learning with Information Sharing","date":"2023-08-16","arxiv_id":"2308.08705","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-robot-challenge-2022-learning-dexterous","title":"Real Robot Challenge 2022: Learning Dexterous Manipulation from Offline Data in the Real World","date":"2023-08-15","arxiv_id":"2308.07741","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-rl-augmented-cold","title":"On-demand Cold Start Frequency Reduction with Off-Policy Reinforcement Learning in Serverless Computing","date":"2023-08-15","arxiv_id":"2308.07541","repositories_listed":0,"syntology":null},{"url":null,"slug":"insurance-pricing-on-price-comparison","title":"Insurance pricing on price comparison websites via reinforcement learning","date":"2023-08-14","arxiv_id":"2308.06935","repositories_listed":0,"syntology":null},{"url":null,"slug":"iob-integrating-optimization-transfer-and","title":"IOB: Integrating Optimization Transfer and Behavior Transfer for Multi-Policy Reuse","date":"2023-08-14","arxiv_id":"2308.07351","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-optimize-lsm-trees-towards-a","title":"Learning to Optimize LSM-trees: Towards A Reinforcement Learning based Key-Value Store for Dynamic Workloads","date":"2023-08-14","arxiv_id":"2308.07013","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-categorical-priors-for-physics-based","title":"Neural Categorical Priors for Physics-Based Character Control","date":"2023-08-14","arxiv_id":"2308.07200","repositories_listed":0,"syntology":null},{"url":null,"slug":"omega-regular-reward-machines","title":"Omega-Regular Reward Machines","date":"2023-08-14","arxiv_id":"2308.07469","repositories_listed":0,"syntology":null},{"url":null,"slug":"intune-reinforcement-learning-based-data","title":"InTune: Reinforcement Learning-based Data Pipeline Optimization for Deep Recommendation Models","date":"2023-08-13","arxiv_id":"2308.08500","repositories_listed":0,"syntology":null},{"url":null,"slug":"cyberforce-a-federated-reinforcement-learning","title":"CyberForce: A Federated Reinforcement Learning Framework for Malware Mitigation","date":"2023-08-11","arxiv_id":"2308.05978","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-classical-and-deep","title":"A Comparison of Classical and Deep Reinforcement Learning Methods for HVAC Control","date":"2023-08-10","arxiv_id":"2308.05711","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-algorithm-for","title":"Provably Efficient Algorithm for Nonstationary Low-Rank MDPs","date":"2023-08-10","arxiv_id":"2308.05471","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-wideband-spectrum-sensing-and","title":"Collaborative Wideband Spectrum Sensing and Scheduling for Networked UAVs in UTM Systems","date":"2023-08-09","arxiv_id":"2308.05036","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-with-variable-time","title":"Actor-Critic with variable time discretization via sustained actions","date":"2023-08-08","arxiv_id":"2308.04299","repositories_listed":0,"syntology":null},{"url":null,"slug":"characterization-of-human-balance-through-a","title":"Characterization of Human Balance through a Reinforcement Learning-based Muscle Controller","date":"2023-08-08","arxiv_id":"2308.04462","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-approach-to-1","title":"A Reinforcement Learning-Based Approach to Graph Discovery in D2D-Enabled Federated Learning","date":"2023-08-07","arxiv_id":"2308.03933","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-generalization-in-offline","title":"Exploiting Generalization in Offline Reinforcement Learning via Unseen State Augmentations","date":"2023-08-07","arxiv_id":"2308.03882","repositories_listed":0,"syntology":null},{"url":null,"slug":"nonprehensile-planar-manipulation-through","title":"Nonprehensile Planar Manipulation through Reinforcement Learning with Multimodal Categorical Exploration","date":"2023-08-04","arxiv_id":"2308.02459","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligning-agent-policy-with-externalities","title":"PARL: A Unified Framework for Policy Alignment in Reinforcement Learning from Human Feedback","date":"2023-08-03","arxiv_id":"2308.02585","repositories_listed":0,"syntology":null},{"url":null,"slug":"bag-of-policies-for-distributional-deep","title":"Bag of Policies for Distributional Deep Exploration","date":"2023-08-03","arxiv_id":"2308.01759","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-gradient-temporal-difference-learning","title":"Revisiting a Design Choice in Gradient Temporal Difference Learning","date":"2023-08-02","arxiv_id":"2308.01170","repositories_listed":0,"syntology":null},{"url":null,"slug":"follow-the-soldiers-with-optimized-single","title":"Follow the Soldiers with Optimized Single-Shot Multibox Detection and Reinforcement Learning","date":"2023-08-02","arxiv_id":"2308.01389","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-generative-ai","title":"Reinforcement Learning for Generative AI: State of the Art, Opportunities and Open Research Challenges","date":"2023-07-31","arxiv_id":"2308.00031","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-under-probabilistic","title":"Reinforcement Learning Under Probabilistic Spatio-Temporal Constraints with Time Windows","date":"2023-07-29","arxiv_id":"2307.15910","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-implicit-behavior-cloning-and-dynamic","title":"Using Implicit Behavior Cloning and Dynamic Movement Primitive to Facilitate Reinforcement Learning for Robot Motion Planning","date":"2023-07-29","arxiv_id":"2307.16062","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialogue-shaping-empowering-agents-through","title":"Dialogue Shaping: Empowering Agents through NPC Interaction","date":"2023-07-28","arxiv_id":"2307.15833","repositories_listed":0,"syntology":null},{"url":null,"slug":"ether-aligning-emergent-communication-for","title":"ETHER: Aligning Emergent Communication for Hindsight Experience Replay","date":"2023-07-28","arxiv_id":"2307.15494","repositories_listed":0,"syntology":null},{"url":null,"slug":"primitive-skill-based-robot-learning-from","title":"Primitive Skill-based Robot Learning from Human Evaluative Feedback","date":"2023-07-28","arxiv_id":"2307.15801","repositories_listed":0,"syntology":null},{"url":null,"slug":"trackagent-6d-object-tracking-via","title":"TrackAgent: 6D Object Tracking via Reinforcement Learning","date":"2023-07-28","arxiv_id":"2307.15671","repositories_listed":0,"syntology":null},{"url":null,"slug":"actions-speak-what-you-want-provably-sample","title":"Actions Speak What You Want: Provably Sample-Efficient Reinforcement Learning of the Quantal Stackelberg Equilibrium from Strategic Feedbacks","date":"2023-07-26","arxiv_id":"2307.14085","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-the-latent-space-of-gans-through","title":"Controlling the Latent Space of GANs through Reinforcement Learning: A Case Study on Task-based Image-to-Image Translation","date":"2023-07-26","arxiv_id":"2307.13978","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-by-guided-safe","title":"Reinforcement Learning by Guided Safe Exploration","date":"2023-07-26","arxiv_id":"2307.14316","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-efficient-orchestrations-for","title":"Communication-Efficient Orchestrations for URLLC Service via Hierarchical Reinforcement Learning","date":"2023-07-25","arxiv_id":"2307.13415","repositories_listed":0,"syntology":null},{"url":null,"slug":"counterfactual-explanation-policies-in-rl","title":"Counterfactual Explanation Policies in RL","date":"2023-07-25","arxiv_id":"2307.13192","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-on-policy","title":"Offline Reinforcement Learning with On-Policy Q-Function Regularization","date":"2023-07-25","arxiv_id":"2307.13824","repositories_listed":0,"syntology":null},{"url":null,"slug":"settling-the-sample-complexity-of-online","title":"Settling the Sample Complexity of Online Reinforcement Learning","date":"2023-07-25","arxiv_id":"2307.13586","repositories_listed":0,"syntology":null},{"url":null,"slug":"structural-credit-assignment-with-coordinated","title":"Structural Credit Assignment with Coordinated Exploration","date":"2023-07-25","arxiv_id":"2307.13256","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-optimal-approximation-factors-in","title":"The Optimal Approximation Factors in Misspecified Off-Policy Value Function Estimation","date":"2023-07-25","arxiv_id":"2307.13332","repositories_listed":0,"syntology":null},{"url":null,"slug":"unbiased-weight-maximization","title":"Unbiased Weight Maximization","date":"2023-07-25","arxiv_id":"2307.13270","repositories_listed":0,"syntology":null},{"url":null,"slug":"exwarp-extrapolation-and-warping-based","title":"ExWarp: Extrapolation and Warping-based Temporal Supersampling for High-frequency Displays","date":"2023-07-24","arxiv_id":"2307.12607","repositories_listed":0,"syntology":null},{"url":null,"slug":"theoretically-guaranteed-policy-improvement","title":"Theoretically Guaranteed Policy Improvement Distilled from Model-Based Planning","date":"2023-07-24","arxiv_id":"2307.12933","repositories_listed":0,"syntology":null},{"url":null,"slug":"dip-rl-demonstration-inferred-preference","title":"DIP-RL: Demonstration-Inferred Preference Learning in Minecraft","date":"2023-07-22","arxiv_id":"2307.12158","repositories_listed":0,"syntology":null},{"url":null,"slug":"game-theoretic-robust-reinforcement-learning","title":"Game-Theoretic Robust Reinforcement Learning Handles Temporally-Coupled Perturbations","date":"2023-07-22","arxiv_id":"2307.12062","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-reality-gap-of-reinforcement","title":"Bridging the Reality Gap of Reinforcement Learning based Traffic Signal Control using Domain Randomization and Meta Learning","date":"2023-07-21","arxiv_id":"2307.11357","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-practical-reinforcement-learning-for","title":"Towards practical reinforcement learning for tokamak magnetic control","date":"2023-07-21","arxiv_id":"2307.11546","repositories_listed":0,"syntology":null},{"url":null,"slug":"reparameterized-policy-learning-for","title":"Reparameterized Policy Learning for Multimodal Trajectory Optimization","date":"2023-07-20","arxiv_id":"2307.10710","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for-vqa","title":"A reinforcement learning approach for VQA validation: an application to diabetic macular edema grading","date":"2023-07-19","arxiv_id":"2307.09886","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-time-reinforcement-learning-new","title":"Continuous-Time Reinforcement Learning: New Design Algorithms with Theoretical Insights and Performance Guarantees","date":"2023-07-18","arxiv_id":"2307.08920","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-cross-segmentation-for-improved","title":"Data Cross-Segmentation for Improved Generalization in Reinforcement Learning Based Algorithmic Trading","date":"2023-07-18","arxiv_id":"2307.09377","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-3d-beam-reforming-for-hovering","title":"Distributed 3D-Beam Reforming for Hovering-Tolerant UAVs Communication over Coexistence: A Deep-Q Learning for Intelligent Space-Air-Ground Integrated Networks","date":"2023-07-18","arxiv_id":"2307.09325","repositories_listed":0,"syntology":null},{"url":null,"slug":"rex-rapid-exploration-and-exploitation-for-ai","title":"REX: Rapid Exploration and eXploitation for AI Agents","date":"2023-07-18","arxiv_id":"2307.08962","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-unified-agent-with-foundation","title":"Towards A Unified Agent with Foundation Models","date":"2023-07-18","arxiv_id":"2307.09668","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-alternative-to-variance-gini-deviation-for","title":"An Alternative to Variance: Gini Deviation for Risk-averse Policy Gradient","date":"2023-07-17","arxiv_id":"2307.08873","repositories_listed":0,"syntology":null},{"url":null,"slug":"basal-bolus-advisor-for-type-1-diabetes-t1d","title":"Basal-Bolus Advisor for Type 1 Diabetes (T1D) Patients Using Multi-Agent Reinforcement Learning (RL) Methodology","date":"2023-07-17","arxiv_id":"2307.08897","repositories_listed":0,"syntology":null},{"url":null,"slug":"quarl-a-learning-based-quantum-circuit","title":"Quarl: A Learning-Based Quantum Circuit Optimizer","date":"2023-07-17","arxiv_id":"2307.10120","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-user-types-mapping-user-traits-by","title":"Discovering User Types: Mapping User Traits by Task-Specific Behaviors in Reinforcement Learning","date":"2023-07-16","arxiv_id":"2307.08169","repositories_listed":0,"syntology":null},{"url":null,"slug":"magnetic-field-based-reward-shaping-for-goal","title":"Magnetic Field-Based Reward Shaping for Goal-Conditioned Reinforcement Learning","date":"2023-07-16","arxiv_id":"2307.08033","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-study-of-the-effectiveness-of","title":"An Empirical Study of the Effectiveness of Using a Replay Buffer on Mode Discovery in GFlowNets","date":"2023-07-15","arxiv_id":"2307.07674","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-model-predictive-control-and","title":"Combining model-predictive control and predictive reinforcement learning for stable quadrupedal robot locomotion","date":"2023-07-15","arxiv_id":"2307.07752","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-action-robust-reinforcement","title":"Efficient Action Robust Reinforcement Learning with Probabilistic Policy Execution Uncertainty","date":"2023-07-15","arxiv_id":"2307.07666","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-is-not-believing-robust-reinforcement","title":"Seeing is not Believing: Robust Reinforcement Learning against Spurious Correlation","date":"2023-07-15","arxiv_id":"2307.07907","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-guided-dialog-policy-learning-performs","title":"Why Guided Dialog Policy Learning performs well? Understanding the role of adversarial learning and its alternative","date":"2023-07-13","arxiv_id":"2307.06721","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-decentralized-partially-observable","title":"Learning Decentralized Partially Observable Mean Field Control for Artificial Collective Behavior","date":"2023-07-12","arxiv_id":"2307.06175","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-in-reinforcement-learning-a","title":"Transformers in Reinforcement Learning: A Survey","date":"2023-07-12","arxiv_id":"2307.05979","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-distributed-multi-task-reinforcement","title":"Scaling Distributed Multi-task Reinforcement Learning with Experience Sharing","date":"2023-07-11","arxiv_id":"2307.05834","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-policies-for-out-of-distribution","title":"Diffusion Policies for Out-of-Distribution Generalization in Offline Reinforcement Learning","date":"2023-07-10","arxiv_id":"2307.04726","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-user-study-on-explainable-online","title":"A User Study on Explainable Online Reinforcement Learning for Adaptive Systems","date":"2023-07-09","arxiv_id":"2307.04098","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-edge-of-stability","title":"Investigating the Edge of Stability Phenomenon in Reinforcement Learning","date":"2023-07-09","arxiv_id":"2307.04210","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-neuromorphic-architecture-for-reinforcement","title":"A Neuromorphic Architecture for Reinforcement Learning from Real-Valued Observations","date":"2023-07-06","arxiv_id":"2307.02947","repositories_listed":0,"syntology":null}],"record_sha256":"809804be530cc76a6164bc04f3bcd783a3c3b4ebf9a6c4354f02165b5f6b6724","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}