{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/69","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":69,"pages_in_order":135,"rows_per_page":100,"rows":[6801,6900],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/68","next":"/task/reinforcement-learning-2/papers/70","papers":[{"url":null,"slug":"can-we-trust-the-evaluation-on-chatgpt","title":"Can we trust the evaluation on ChatGPT?","date":"2023-03-22","arxiv_id":"2303.12767","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-load-balancing-via-efficient","title":"Communication Load Balancing via Efficient Inverse Reinforcement Learning","date":"2023-03-22","arxiv_id":"2303.16686","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-2","title":"Deep Reinforcement Learning for Localizability-Enhanced Navigation in Dynamic Human Environments","date":"2023-03-22","arxiv_id":"2303.12354","repositories_listed":0,"syntology":null},{"url":null,"slug":"haps-uav-enabled-heterogeneous-networks-a","title":"HAPS-UAV-Enabled Heterogeneous Networks: A Deep Reinforcement Learning Approach","date":"2023-03-22","arxiv_id":"2303.12883","repositories_listed":0,"syntology":null},{"url":null,"slug":"p-3-o-transferring-visual-representations-for","title":"$P^{3}O$: Transferring Visual Representations for Reinforcement Learning via Prompting","date":"2023-03-22","arxiv_id":"2303.12371","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-exogenous-states","title":"Reinforcement Learning with Exogenous States and Rewards","date":"2023-03-22","arxiv_id":"2303.12957","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategy-synthesis-in-markov-decision","title":"Strategy Synthesis in Markov Decision Processes Under Limited Sampling Access","date":"2023-03-22","arxiv_id":"2303.12718","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-16","title":"Large-Scale Traffic Signal Control Using Constrained Network Partition and Adaptive Deep Reinforcement Learning","date":"2023-03-21","arxiv_id":"2303.11899","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-imitation-and-online-reinforcement","title":"Bridging Imitation and Online Reinforcement Learning: An Optimistic Tale","date":"2023-03-20","arxiv_id":"2303.11369","repositories_listed":0,"syntology":null},{"url":null,"slug":"deceptive-reinforcement-learning-in-model","title":"Deceptive Reinforcement Learning in Model-Free Domains","date":"2023-03-20","arxiv_id":"2303.10838","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-sample-complexity-for-reward-free","title":"Improved Sample Complexity for Reward-free Reinforcement Learning under Low-rank MDPs","date":"2023-03-20","arxiv_id":"2303.10859","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-hypothesis-testing-in-unknown","title":"Active hypothesis testing in unknown environments using recurrent neural networks and model free reinforcement learning","date":"2023-03-19","arxiv_id":"2303.10623","repositories_listed":0,"syntology":null},{"url":null,"slug":"boundary-aware-supervoxel-level-iteratively","title":"Boundary-aware Supervoxel-level Iteratively Refined Interactive 3D Image Segmentation with Multi-agent Reinforcement Learning","date":"2023-03-19","arxiv_id":"2303.10692","repositories_listed":0,"syntology":null},{"url":null,"slug":"cheap-talk-discovery-and-utilization-in-multi","title":"Cheap Talk Discovery and Utilization in Multi-Agent Reinforcement Learning","date":"2023-03-19","arxiv_id":"2303.10733","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-via-mean","title":"Major-Minor Mean Field Multi-Agent Reinforcement Learning","date":"2023-03-19","arxiv_id":"2303.10665","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-reward-for-visual-relationships","title":"Multi-modal reward for visual relationships-based image captioning","date":"2023-03-19","arxiv_id":"2303.10766","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-reinforcement-learning-via-1","title":"Interpretable Reinforcement Learning via Neural Additive Models for Inventory Management","date":"2023-03-18","arxiv_id":"2303.10382","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-data-driven-model-reference-adaptive","title":"A Data-Driven Model-Reference Adaptive Control Approach Based on Reinforcement Learning","date":"2023-03-17","arxiv_id":"2303.09994","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-policy-iteration-algorithm-for","title":"A New Policy Iteration Algorithm For Reinforcement Learning in Zero-Sum Markov Games","date":"2023-03-17","arxiv_id":"2303.09716","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-nars-and-reinforcement-learning-an","title":"Comparing NARS and Reinforcement Learning: An Analysis of ONA and $Q$-Learning Algorithms","date":"2023-03-17","arxiv_id":"2304.03291","repositories_listed":0,"syntology":null},{"url":null,"slug":"measurement-optimization-under-uncertainty","title":"Measurement Optimization under Uncertainty using Deep Reinforcement Learning","date":"2023-03-17","arxiv_id":"2303.09750","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-safe-propofol-dosing-during-general","title":"Towards Real-World Applications of Personalized Anesthesia Using Policy Constraint Q Learning for Propofol Infusion Control","date":"2023-03-17","arxiv_id":"2303.10180","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-conditioned-offline-reinforcement","title":"Goal-conditioned Offline Reinforcement Learning through State Space Partitioning","date":"2023-03-16","arxiv_id":"2303.09367","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-rewards-to-optimize-global","title":"Learning Rewards to Optimize Global Performance Metrics in Deep Reinforcement Learning","date":"2023-03-16","arxiv_id":"2303.09027","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-reinforcement-learning-in-periodic-mdp","title":"Online Reinforcement Learning in Periodic MDP","date":"2023-03-16","arxiv_id":"2303.09629","repositories_listed":0,"syntology":null},{"url":null,"slug":"psychotherapy-ai-companion-with-reinforcement","title":"Psychotherapy AI Companion with Reinforcement Learning Recommendations and Interpretable Policy Dynamics","date":"2023-03-16","arxiv_id":"2303.09601","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-omega-regular","title":"Reinforcement Learning for Omega-Regular Specifications on Continuous-Time MDP","date":"2023-03-16","arxiv_id":"2303.09528","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-inspection-method-of-unmanned-aerial","title":"Self-Inspection Method of Unmanned Aerial Vehicles in Power Plants Using Deep Q-Network Reinforcement Learning","date":"2023-03-16","arxiv_id":"2303.09013","repositories_listed":0,"syntology":null},{"url":null,"slug":"svde-scalable-value-decomposition-exploration","title":"SVDE: Scalable Value-Decomposition Exploration for Cooperative Multi-Agent Reinforcement Learning","date":"2023-03-16","arxiv_id":"2303.09058","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-conditioned-policy-gradient-for-multi","title":"Latent-Conditioned Policy Gradient for Multi-Objective Deep Reinforcement Learning","date":"2023-03-15","arxiv_id":"2303.08909","repositories_listed":0,"syntology":null},{"url":null,"slug":"muti-agent-proximal-policy-optimization-for","title":"Muti-Agent Proximal Policy Optimization For Data Freshness in UAV-assisted Networks","date":"2023-03-15","arxiv_id":"2303.08680","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-measurement-driven-reinforcement","title":"Real-Time Measurement-Driven Reinforcement Learning Control Approach for Uncertain Nonlinear Systems","date":"2023-03-15","arxiv_id":"2303.08745","repositories_listed":0,"syntology":null},{"url":null,"slug":"replay-buffer-with-local-forgetting-for","title":"Replay Buffer with Local Forgetting for Adapting to Local Environment Changes in Deep Model-Based Reinforcement Learning","date":"2023-03-15","arxiv_id":"2303.08690","repositories_listed":0,"syntology":null},{"url":null,"slug":"smoothed-q-learning","title":"Smoothed Q-learning","date":"2023-03-15","arxiv_id":"2303.08631","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategic-trading-in-quantitative-markets","title":"Optimizing Trading Strategies in Quantitative Markets using Multi-Agent Reinforcement Learning","date":"2023-03-15","arxiv_id":"2303.11959","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-policy-learning-for-offline-to","title":"Adaptive Policy Learning for Offline-to-Online Reinforcement Learning","date":"2023-03-14","arxiv_id":"2303.07693","repositories_listed":0,"syntology":null},{"url":null,"slug":"bi-directional-personalization-reinforcement","title":"Bi-directional personalization reinforcement learning-based architecture with active learning using a multi-model data service for the travel nursing industry","date":"2023-03-14","arxiv_id":"2304.00006","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-learning-for-mean-field-control","title":"Actor-Critic learning for mean-field control in continuous time","date":"2023-03-13","arxiv_id":"2303.06993","repositories_listed":0,"syntology":null},{"url":null,"slug":"deploying-offline-reinforcement-learning-with","title":"Deploying Offline Reinforcement Learning with Human Feedback","date":"2023-03-13","arxiv_id":"2303.07046","repositories_listed":0,"syntology":null},{"url":null,"slug":"loss-of-plasticity-in-continual-deep","title":"Loss of Plasticity in Continual Deep Reinforcement Learning","date":"2023-03-13","arxiv_id":"2303.07507","repositories_listed":0,"syntology":null},{"url":null,"slug":"path-planning-using-reinforcement-learning-a","title":"Path Planning using Reinforcement Learning: A Policy Iteration Approach","date":"2023-03-13","arxiv_id":"2303.07535","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-tree-reconstruction-game-phylogenetic","title":"The tree reconstruction game: phylogenetic reconstruction using reinforcement learning","date":"2023-03-12","arxiv_id":"2303.06695","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-synergies-between-quality","title":"Understanding the Synergies between Quality-Diversity and Deep Reinforcement Learning","date":"2023-03-10","arxiv_id":"2303.06164","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-history-aware-hyperparameter","title":"A Framework for History-Aware Hyperparameter Optimisation in Reinforcement Learning","date":"2023-03-09","arxiv_id":"2303.05186","repositories_listed":0,"syntology":null},{"url":null,"slug":"beware-of-instantaneous-dependence-in","title":"Beware of Instantaneous Dependence in Reinforcement Learning","date":"2023-03-09","arxiv_id":"2303.05458","repositories_listed":0,"syntology":null},{"url":null,"slug":"computably-continuous-reinforcement-learning","title":"Computably Continuous Reinforcement-Learning Objectives are PAC-learnable","date":"2023-03-09","arxiv_id":"2303.05518","repositories_listed":0,"syntology":null},{"url":null,"slug":"conceptual-reinforcement-learning-for","title":"Conceptual Reinforcement Learning for Language-Conditioned Tasks","date":"2023-03-09","arxiv_id":"2303.05069","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-contextual-structure-to-generate","title":"Exploiting Contextual Structure to Generate Useful Auxiliary Tasks","date":"2023-03-09","arxiv_id":"2303.05038","repositories_listed":0,"syntology":null},{"url":null,"slug":"goats-goal-sampling-adaptation-for-scooping","title":"GOATS: Goal Sampling Adaptation for Scooping with Curriculum Reinforcement Learning","date":"2023-03-09","arxiv_id":"2303.05193","repositories_listed":0,"syntology":null},{"url":null,"slug":"power-and-interference-control-for-vlc-based","title":"Power and Interference Control for VLC-Based UDN: A Reinforcement Learning Approach","date":"2023-03-09","arxiv_id":"2303.05448","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-scheduling-of-renewable-power","title":"Real-time scheduling of renewable power systems through planning-based reinforcement learning","date":"2023-03-09","arxiv_id":"2303.05205","repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-advances-of-deep-robotic-affordance","title":"Recent Advances of Deep Robotic Affordance Learning: A Reinforcement Learning Perspective","date":"2023-03-09","arxiv_id":"2303.05344","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-informed-dreamer-for-task","title":"Task Aware Dreamer for Task Generalization in Reinforcement Learning","date":"2023-03-09","arxiv_id":"2303.05092","repositories_listed":0,"syntology":null},{"url":null,"slug":"variance-aware-robust-reinforcement-learning","title":"Variance-aware robust reinforcement learning with linear function approximation under heavy-tailed rewards","date":"2023-03-09","arxiv_id":"2303.05606","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-memory-based-learning-to-solve-tasks","title":"Using Memory-Based Learning to Solve Tasks with State-Action Constraints","date":"2023-03-08","arxiv_id":"2303.04327","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-occupancy-predictive-representations-for","title":"Deep Occupancy-Predictive Representations for Autonomous Driving","date":"2023-03-07","arxiv_id":"2303.04218","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-randomization-for-robust-affordable","title":"Domain Randomization for Robust, Affordable and Effective Closed-loop Control of Soft Robots","date":"2023-03-07","arxiv_id":"2303.04136","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-environment-transformer-and-offline","title":"Environment Transformer and Policy Optimization for Model-Based Offline Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03811","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-reinforcement-learning-a-survey","title":"Evolutionary Reinforcement Learning: A Survey","date":"2023-03-07","arxiv_id":"2303.04150","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-sample-complexity-of-vanilla-model","title":"On the Sample Complexity of Vanilla Model-Based Offline Reinforcement Learning with Dependent Samples","date":"2023-03-07","arxiv_id":"2303.04268","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-humanoid-locomotion-with","title":"Real-World Humanoid Locomotion with Reinforcement Learning","date":"2023-03-06","arxiv_id":"2303.03381","repositories_listed":0,"syntology":null},{"url":null,"slug":"maestro-open-ended-environment-design-for","title":"MAESTRO: Open-Ended Environment Design for Multi-Agent Reinforcement Learning","date":"2023-03-06","arxiv_id":"2303.03376","repositories_listed":0,"syntology":null},{"url":null,"slug":"perspectives-on-the-social-impacts-of","title":"Perspectives on the Social Impacts of Reinforcement Learning with Human Feedback","date":"2023-03-06","arxiv_id":"2303.02891","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-guided-exploration-with-sub-optimal","title":"Dexterous In-hand Manipulation by Guiding Exploration with Simple Sub-skill Controllers","date":"2023-03-06","arxiv_id":"2303.03533","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-reinforcement-learning-a-survey","title":"Ensemble Reinforcement Learning: A Survey","date":"2023-03-05","arxiv_id":"2303.02618","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-environment-poisoning-attacks-on","title":"Local Environment Poisoning Attacks on Federated Reinforcement Learning","date":"2023-03-05","arxiv_id":"2303.02725","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-a3c-deep-reinforcement-learning-on","title":"Double A3C: Deep Reinforcement Learning on OpenAI Gym Games","date":"2023-03-04","arxiv_id":"2303.02271","repositories_listed":0,"syntology":null},{"url":null,"slug":"look-ahead-ac-optimal-power-flow-a-model","title":"Look-Ahead AC Optimal Power Flow: A Model-Informed Reinforcement Learning Approach","date":"2023-03-04","arxiv_id":"2303.02306","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximating-energy-market-clearing-and","title":"Approximating Energy Market Clearing and Bidding With Model-Based Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.01772","repositories_listed":0,"syntology":null},{"url":null,"slug":"guarded-policy-optimization-with-imperfect","title":"Guarded Policy Optimization with Imperfect Online Demonstrations","date":"2023-03-03","arxiv_id":"2303.01728","repositories_listed":0,"syntology":null},{"url":null,"slug":"hindsight-states-blending-sim-and-real-task","title":"Hindsight States: Blending Sim and Real Task Elements for Efficient Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.02234","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-o-ran-traffic-steering-for-urllc","title":"Intelligent O-RAN Traffic Steering for URLLC Through Deep Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.01960","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-influence-human-behavior-with","title":"Learning to Influence Human Behavior with Offline Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.02265","repositories_listed":0,"syntology":null},{"url":null,"slug":"reprem-representation-pre-training-with","title":"RePreM: Representation Pre-training with Masked Model for Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.01668","repositories_listed":0,"syntology":null},{"url":null,"slug":"tile-networks-learning-optimal-geometric","title":"Tile Networks: Learning Optimal Geometric Layout for Whole-page Recommendation","date":"2023-03-03","arxiv_id":"2303.01671","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-risk-based-optimistic-exploration-for","title":"Toward Risk-based Optimistic Exploration for Cooperative Multi-Agent Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.01768","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-learning-planning-and-control-policies","title":"Co-learning Planning and Control Policies Constrained by Differentiable Logic Specifications","date":"2023-03-02","arxiv_id":"2303.01346","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adaptation-of-reinforcement-learning","title":"Domain Adaptation of Reinforcement Learning Agents based on Network Service Proximity","date":"2023-03-02","arxiv_id":"2303.01013","repositories_listed":0,"syntology":null},{"url":null,"slug":"expert-free-online-transfer-learning-in-multi","title":"Expert-Free Online Transfer Learning in Multi-Agent Reinforcement Learning","date":"2023-03-02","arxiv_id":"2303.01170","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-sharing-with-network-pruning-for","title":"Parameter Sharing with Network Pruning for Scalable Multi-Agent Deep Reinforcement Learning","date":"2023-03-02","arxiv_id":"2303.00912","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-labels-multi-agent-deep","title":"Reinforced Labels: Multi-Agent Deep Reinforcement Learning for Point-Feature Label Placement","date":"2023-03-02","arxiv_id":"2303.01388","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-constrained-station-keeping-for","title":"Resource-Constrained Station-Keeping for Helium Balloons using Reinforcement Learning","date":"2023-03-02","arxiv_id":"2303.01173","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-improving-robots-end-to-end-autonomous","title":"Self-Improving Robots: End-to-End Autonomous Visuomotor Reinforcement Learning","date":"2023-03-02","arxiv_id":"2303.01488","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-trader-without","title":"A Deep Reinforcement Learning Trader without Offline Training","date":"2023-03-01","arxiv_id":"2303.00356","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-variational-approach-to-mutual-information","title":"A Variational Approach to Mutual Information-Based Coordination for Multi-Agent Reinforcement Learning","date":"2023-03-01","arxiv_id":"2303.00451","repositories_listed":0,"syntology":null},{"url":null,"slug":"ar3n-a-reinforcement-learning-based-assist-as","title":"AR3n: A Reinforcement Learning-based Assist-As-Needed Controller for Robotic Rehabilitation","date":"2023-02-28","arxiv_id":"2303.00085","repositories_listed":0,"syntology":null},{"url":null,"slug":"auxiliary-task-based-deep-reinforcement-1","title":"Auxiliary Task-based Deep Reinforcement Learning for Quantum Control","date":"2023-02-28","arxiv_id":"2302.14312","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-reinforcement-learning-for-operator","title":"Graph Reinforcement Learning for Operator Selection in the ALNS Metaheuristic","date":"2023-02-28","arxiv_id":"2302.14678","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-in","title":"Hierarchical Reinforcement Learning in Complex 3D Environments","date":"2023-02-28","arxiv_id":"2302.14451","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimizing-the-outage-probability-in-a-markov","title":"Minimizing the Outage Probability in a Markov Decision Process","date":"2023-02-28","arxiv_id":"2302.14714","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-15","title":"Multi-Agent Reinforcement Learning for Pragmatic Communication and Control","date":"2023-02-28","arxiv_id":"2302.14399","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-optimization-of-llc-converter-with","title":"Parameter Optimization of LLC-Converter with multiple operation points using Reinforcement Learning","date":"2023-02-28","arxiv_id":"2303.00004","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for-7","title":"A Reinforcement Learning Approach for Scheduling Problems With Improved Generalization Through Order Swapping","date":"2023-02-27","arxiv_id":"2302.13941","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-method-for-risk-averse","title":"Distributional Method for Risk Averse Reinforcement Learning","date":"2023-02-27","arxiv_id":"2302.14109","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-resource-allocation-for-metaverse","title":"Dynamic Resource Allocation for Metaverse Applications with Deep Reinforcement Learning","date":"2023-02-27","arxiv_id":"2302.13445","repositories_listed":0,"syntology":null},{"url":null,"slug":"exposure-based-multi-agent-inspection-of-a","title":"Exposure-Based Multi-Agent Inspection of a Tumbling Target Using Deep Reinforcement Learning","date":"2023-02-27","arxiv_id":"2302.14188","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-depreciating","title":"Reinforcement Learning with Depreciating Assets","date":"2023-02-27","arxiv_id":"2302.14176","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-provable-benefits-of-unsupervised-data","title":"The Provable Benefits of Unsupervised Data Sharing for Offline Reinforcement Learning","date":"2023-02-27","arxiv_id":"2302.13493","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-cogni-an-integrated-causal-reinforcement","title":"Q-Cogni: An Integrated Causal Reinforcement Learning Framework","date":"2023-02-26","arxiv_id":"2302.13240","repositories_listed":0,"syntology":null},{"url":null,"slug":"revolutionizing-genomics-with-reinforcement","title":"Revolutionizing Genomics with Reinforcement Learning Techniques","date":"2023-02-26","arxiv_id":"2302.13268","repositories_listed":0,"syntology":null}],"record_sha256":"4dd794b8fc1e93c24e1d23a696cfe91448f6a7f4021330b741dc827fbf3f44c0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}