{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/66","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":66,"pages_in_order":132,"rows_per_page":100,"rows":[6501,6600],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/65","next":"/task/reinforcement-learning/papers/67","papers":[{"url":null,"slug":"learning-generative-models-with-goal","title":"Learning Generative Models with Goal-conditioned Reinforcement Learning","date":"2023-03-26","arxiv_id":"2303.14811","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-packaging-optimization-with","title":"Robotic Packaging Optimization with Reinforcement Learning","date":"2023-03-26","arxiv_id":"2303.14693","repositories_listed":0,"syntology":null},{"url":null,"slug":"causality-detection-for-efficient-multi-agent","title":"Causality Detection for Efficient Multi-Agent Reinforcement Learning","date":"2023-03-24","arxiv_id":"2303.14227","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-path-following-on-rivers-using","title":"Robust Path Following on Rivers Using Bootstrapped Reinforcement Learning","date":"2023-03-24","arxiv_id":"2303.15178","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-knockoffs-for-variable-selection","title":"Sequential Knockoffs for Variable Selection in Reinforcement Learning","date":"2023-03-24","arxiv_id":"2303.14281","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-reinforcement-learning-and-planning","title":"Boosting Reinforcement Learning and Planning with Demonstrations: A Survey","date":"2023-03-23","arxiv_id":"2303.13489","repositories_listed":0,"syntology":null},{"url":null,"slug":"connected-superlevel-set-in-deep","title":"Connected Superlevel Set in (Deep) Reinforcement Learning and its Application to Minimax Theorems","date":"2023-03-23","arxiv_id":"2303.12981","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-load-balancing-via-efficient","title":"Communication Load Balancing via Efficient Inverse Reinforcement Learning","date":"2023-03-22","arxiv_id":"2303.16686","repositories_listed":0,"syntology":null},{"url":null,"slug":"haps-uav-enabled-heterogeneous-networks-a","title":"HAPS-UAV-Enabled Heterogeneous Networks: A Deep Reinforcement Learning Approach","date":"2023-03-22","arxiv_id":"2303.12883","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-exogenous-states","title":"Reinforcement Learning with Exogenous States and Rewards","date":"2023-03-22","arxiv_id":"2303.12957","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategy-synthesis-in-markov-decision","title":"Strategy Synthesis in Markov Decision Processes Under Limited Sampling Access","date":"2023-03-22","arxiv_id":"2303.12718","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-imitation-and-online-reinforcement","title":"Bridging Imitation and Online Reinforcement Learning: An Optimistic Tale","date":"2023-03-20","arxiv_id":"2303.11369","repositories_listed":0,"syntology":null},{"url":null,"slug":"deceptive-reinforcement-learning-in-model","title":"Deceptive Reinforcement Learning in Model-Free Domains","date":"2023-03-20","arxiv_id":"2303.10838","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-hypothesis-testing-in-unknown","title":"Active hypothesis testing in unknown environments using recurrent neural networks and model free reinforcement learning","date":"2023-03-19","arxiv_id":"2303.10623","repositories_listed":0,"syntology":null},{"url":null,"slug":"boundary-aware-supervoxel-level-iteratively","title":"Boundary-aware Supervoxel-level Iteratively Refined Interactive 3D Image Segmentation with Multi-agent Reinforcement Learning","date":"2023-03-19","arxiv_id":"2303.10692","repositories_listed":0,"syntology":null},{"url":null,"slug":"cheap-talk-discovery-and-utilization-in-multi","title":"Cheap Talk Discovery and Utilization in Multi-Agent Reinforcement Learning","date":"2023-03-19","arxiv_id":"2303.10733","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-via-mean","title":"Major-Minor Mean Field Multi-Agent Reinforcement Learning","date":"2023-03-19","arxiv_id":"2303.10665","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-reinforcement-learning-via-1","title":"Interpretable Reinforcement Learning via Neural Additive Models for Inventory Management","date":"2023-03-18","arxiv_id":"2303.10382","repositories_listed":0,"syntology":null},{"url":null,"slug":"measurement-optimization-under-uncertainty","title":"Measurement Optimization under Uncertainty using Deep Reinforcement Learning","date":"2023-03-17","arxiv_id":"2303.09750","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-conditioned-offline-reinforcement","title":"Goal-conditioned Offline Reinforcement Learning through State Space Partitioning","date":"2023-03-16","arxiv_id":"2303.09367","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-reinforcement-learning-in-periodic-mdp","title":"Online Reinforcement Learning in Periodic MDP","date":"2023-03-16","arxiv_id":"2303.09629","repositories_listed":0,"syntology":null},{"url":null,"slug":"psychotherapy-ai-companion-with-reinforcement","title":"Psychotherapy AI Companion with Reinforcement Learning Recommendations and Interpretable Policy Dynamics","date":"2023-03-16","arxiv_id":"2303.09601","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-omega-regular","title":"Reinforcement Learning for Omega-Regular Specifications on Continuous-Time MDP","date":"2023-03-16","arxiv_id":"2303.09528","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-inspection-method-of-unmanned-aerial","title":"Self-Inspection Method of Unmanned Aerial Vehicles in Power Plants Using Deep Q-Network Reinforcement Learning","date":"2023-03-16","arxiv_id":"2303.09013","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-conditioned-policy-gradient-for-multi","title":"Latent-Conditioned Policy Gradient for Multi-Objective Deep Reinforcement Learning","date":"2023-03-15","arxiv_id":"2303.08909","repositories_listed":0,"syntology":null},{"url":null,"slug":"muti-agent-proximal-policy-optimization-for","title":"Muti-Agent Proximal Policy Optimization For Data Freshness in UAV-assisted Networks","date":"2023-03-15","arxiv_id":"2303.08680","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-measurement-driven-reinforcement","title":"Real-Time Measurement-Driven Reinforcement Learning Control Approach for Uncertain Nonlinear Systems","date":"2023-03-15","arxiv_id":"2303.08745","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-policy-learning-for-offline-to","title":"Adaptive Policy Learning for Offline-to-Online Reinforcement Learning","date":"2023-03-14","arxiv_id":"2303.07693","repositories_listed":0,"syntology":null},{"url":null,"slug":"bi-directional-personalization-reinforcement","title":"Bi-directional personalization reinforcement learning-based architecture with active learning using a multi-model data service for the travel nursing industry","date":"2023-03-14","arxiv_id":"2304.00006","repositories_listed":0,"syntology":null},{"url":null,"slug":"deploying-offline-reinforcement-learning-with","title":"Deploying Offline Reinforcement Learning with Human Feedback","date":"2023-03-13","arxiv_id":"2303.07046","repositories_listed":0,"syntology":null},{"url":null,"slug":"loss-of-plasticity-in-continual-deep","title":"Loss of Plasticity in Continual Deep Reinforcement Learning","date":"2023-03-13","arxiv_id":"2303.07507","repositories_listed":0,"syntology":null},{"url":null,"slug":"path-planning-using-reinforcement-learning-a","title":"Path Planning using Reinforcement Learning: A Policy Iteration Approach","date":"2023-03-13","arxiv_id":"2303.07535","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-tree-reconstruction-game-phylogenetic","title":"The tree reconstruction game: phylogenetic reconstruction using reinforcement learning","date":"2023-03-12","arxiv_id":"2303.06695","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-synergies-between-quality","title":"Understanding the Synergies between Quality-Diversity and Deep Reinforcement Learning","date":"2023-03-10","arxiv_id":"2303.06164","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-history-aware-hyperparameter","title":"A Framework for History-Aware Hyperparameter Optimisation in Reinforcement Learning","date":"2023-03-09","arxiv_id":"2303.05186","repositories_listed":0,"syntology":null},{"url":null,"slug":"beware-of-instantaneous-dependence-in","title":"Beware of Instantaneous Dependence in Reinforcement Learning","date":"2023-03-09","arxiv_id":"2303.05458","repositories_listed":0,"syntology":null},{"url":null,"slug":"computably-continuous-reinforcement-learning","title":"Computably Continuous Reinforcement-Learning Objectives are PAC-learnable","date":"2023-03-09","arxiv_id":"2303.05518","repositories_listed":0,"syntology":null},{"url":null,"slug":"conceptual-reinforcement-learning-for","title":"Conceptual Reinforcement Learning for Language-Conditioned Tasks","date":"2023-03-09","arxiv_id":"2303.05069","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-contextual-structure-to-generate","title":"Exploiting Contextual Structure to Generate Useful Auxiliary Tasks","date":"2023-03-09","arxiv_id":"2303.05038","repositories_listed":0,"syntology":null},{"url":null,"slug":"goats-goal-sampling-adaptation-for-scooping","title":"GOATS: Goal Sampling Adaptation for Scooping with Curriculum Reinforcement Learning","date":"2023-03-09","arxiv_id":"2303.05193","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-scheduling-of-renewable-power","title":"Real-time scheduling of renewable power systems through planning-based reinforcement learning","date":"2023-03-09","arxiv_id":"2303.05205","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-informed-dreamer-for-task","title":"Task Aware Dreamer for Task Generalization in Reinforcement Learning","date":"2023-03-09","arxiv_id":"2303.05092","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-memory-based-learning-to-solve-tasks","title":"Using Memory-Based Learning to Solve Tasks with State-Action Constraints","date":"2023-03-08","arxiv_id":"2303.04327","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-randomization-for-robust-affordable","title":"Domain Randomization for Robust, Affordable and Effective Closed-loop Control of Soft Robots","date":"2023-03-07","arxiv_id":"2303.04136","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-reinforcement-learning-a-survey","title":"Evolutionary Reinforcement Learning: A Survey","date":"2023-03-07","arxiv_id":"2303.04150","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-sample-complexity-of-vanilla-model","title":"On the Sample Complexity of Vanilla Model-Based Offline Reinforcement Learning with Dependent Samples","date":"2023-03-07","arxiv_id":"2303.04268","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-humanoid-locomotion-with","title":"Real-World Humanoid Locomotion with Reinforcement Learning","date":"2023-03-06","arxiv_id":"2303.03381","repositories_listed":0,"syntology":null},{"url":null,"slug":"maestro-open-ended-environment-design-for","title":"MAESTRO: Open-Ended Environment Design for Multi-Agent Reinforcement Learning","date":"2023-03-06","arxiv_id":"2303.03376","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-reinforcement-learning-a-survey","title":"Ensemble Reinforcement Learning: A Survey","date":"2023-03-05","arxiv_id":"2303.02618","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-environment-poisoning-attacks-on","title":"Local Environment Poisoning Attacks on Federated Reinforcement Learning","date":"2023-03-05","arxiv_id":"2303.02725","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-a3c-deep-reinforcement-learning-on","title":"Double A3C: Deep Reinforcement Learning on OpenAI Gym Games","date":"2023-03-04","arxiv_id":"2303.02271","repositories_listed":0,"syntology":null},{"url":null,"slug":"look-ahead-ac-optimal-power-flow-a-model","title":"Look-Ahead AC Optimal Power Flow: A Model-Informed Reinforcement Learning Approach","date":"2023-03-04","arxiv_id":"2303.02306","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximating-energy-market-clearing-and","title":"Approximating Energy Market Clearing and Bidding With Model-Based Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.01772","repositories_listed":0,"syntology":null},{"url":null,"slug":"guarded-policy-optimization-with-imperfect","title":"Guarded Policy Optimization with Imperfect Online Demonstrations","date":"2023-03-03","arxiv_id":"2303.01728","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-influence-human-behavior-with","title":"Learning to Influence Human Behavior with Offline Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.02265","repositories_listed":0,"syntology":null},{"url":null,"slug":"reprem-representation-pre-training-with","title":"RePreM: Representation Pre-training with Masked Model for Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.01668","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-risk-based-optimistic-exploration-for","title":"Toward Risk-based Optimistic Exploration for Cooperative Multi-Agent Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.01768","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-learning-planning-and-control-policies","title":"Co-learning Planning and Control Policies Constrained by Differentiable Logic Specifications","date":"2023-03-02","arxiv_id":"2303.01346","repositories_listed":0,"syntology":null},{"url":null,"slug":"expert-free-online-transfer-learning-in-multi","title":"Expert-Free Online Transfer Learning in Multi-Agent Reinforcement Learning","date":"2023-03-02","arxiv_id":"2303.01170","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-labels-multi-agent-deep","title":"Reinforced Labels: Multi-Agent Deep Reinforcement Learning for Point-Feature Label Placement","date":"2023-03-02","arxiv_id":"2303.01388","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-constrained-station-keeping-for","title":"Resource-Constrained Station-Keeping for Helium Balloons using Reinforcement Learning","date":"2023-03-02","arxiv_id":"2303.01173","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-improving-robots-end-to-end-autonomous","title":"Self-Improving Robots: End-to-End Autonomous Visuomotor Reinforcement Learning","date":"2023-03-02","arxiv_id":"2303.01488","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-trader-without","title":"A Deep Reinforcement Learning Trader without Offline Training","date":"2023-03-01","arxiv_id":"2303.00356","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-variational-approach-to-mutual-information","title":"A Variational Approach to Mutual Information-Based Coordination for Multi-Agent Reinforcement Learning","date":"2023-03-01","arxiv_id":"2303.00451","repositories_listed":0,"syntology":null},{"url":null,"slug":"auxiliary-task-based-deep-reinforcement-1","title":"Auxiliary Task-based Deep Reinforcement Learning for Quantum Control","date":"2023-02-28","arxiv_id":"2302.14312","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-reinforcement-learning-for-operator","title":"Graph Reinforcement Learning for Operator Selection in the ALNS Metaheuristic","date":"2023-02-28","arxiv_id":"2302.14678","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-in","title":"Hierarchical Reinforcement Learning in Complex 3D Environments","date":"2023-02-28","arxiv_id":"2302.14451","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimizing-the-outage-probability-in-a-markov","title":"Minimizing the Outage Probability in a Markov Decision Process","date":"2023-02-28","arxiv_id":"2302.14714","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-15","title":"Multi-Agent Reinforcement Learning for Pragmatic Communication and Control","date":"2023-02-28","arxiv_id":"2302.14399","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-method-for-risk-averse","title":"Distributional Method for Risk Averse Reinforcement Learning","date":"2023-02-27","arxiv_id":"2302.14109","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-resource-allocation-for-metaverse","title":"Dynamic Resource Allocation for Metaverse Applications with Deep Reinforcement Learning","date":"2023-02-27","arxiv_id":"2302.13445","repositories_listed":0,"syntology":null},{"url":null,"slug":"exposure-based-multi-agent-inspection-of-a","title":"Exposure-Based Multi-Agent Inspection of a Tumbling Target Using Deep Reinforcement Learning","date":"2023-02-27","arxiv_id":"2302.14188","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-depreciating","title":"Reinforcement Learning with Depreciating Assets","date":"2023-02-27","arxiv_id":"2302.14176","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-cogni-an-integrated-causal-reinforcement","title":"Q-Cogni: An Integrated Causal Reinforcement Learning Framework","date":"2023-02-26","arxiv_id":"2302.13240","repositories_listed":0,"syntology":null},{"url":null,"slug":"revolutionizing-genomics-with-reinforcement","title":"Revolutionizing Genomics with Reinforcement Learning Techniques","date":"2023-02-26","arxiv_id":"2302.13268","repositories_listed":0,"syntology":null},{"url":null,"slug":"exponential-hardness-of-reinforcement","title":"Exponential Hardness of Reinforcement Learning with Linear Function Approximation","date":"2023-02-25","arxiv_id":"2302.12940","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-bellman-s-principle-of-optimality-and","title":"On Bellman's principle of optimality and Reinforcement learning for safety-constrained Markov decision process","date":"2023-02-25","arxiv_id":"2302.13152","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-jumpy-models-for-planning-and-fast","title":"Leveraging Jumpy Models for Planning and Fast Learning in Robotic Domains","date":"2023-02-24","arxiv_id":"2302.12617","repositories_listed":0,"syntology":null},{"url":null,"slug":"logarithmic-switching-cost-in-reinforcement","title":"Logarithmic Switching Cost in Reinforcement Learning beyond Linear MDPs","date":"2023-02-24","arxiv_id":"2302.12456","repositories_listed":0,"syntology":null},{"url":null,"slug":"concept-learning-for-interpretable-multi","title":"Concept Learning for Interpretable Multi-Agent Reinforcement Learning","date":"2023-02-23","arxiv_id":"2302.12232","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-reinforcement-learning-via","title":"Provably Efficient Reinforcement Learning via Surprise Bound","date":"2023-02-22","arxiv_id":"2302.11634","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-framework-for-online","title":"A Reinforcement Learning Framework for Online Speaker Diarization","date":"2023-02-21","arxiv_id":"2302.10924","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-model-for-offline-reinforcement","title":"Adversarial Model for Offline Reinforcement Learning","date":"2023-02-21","arxiv_id":"2302.11048","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditioning-hierarchical-reinforcement","title":"Handling Long and Richly Constrained Tasks through Constrained Hierarchical Reinforcement Learning","date":"2023-02-21","arxiv_id":"2302.10639","repositories_listed":0,"syntology":null},{"url":null,"slug":"curiosity-driven-exploration-in-sparse-reward","title":"Curiosity-driven Exploration in Sparse-reward Multi-agent Reinforcement Learning","date":"2023-02-21","arxiv_id":"2302.10825","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-for-mixture-of","title":"Offline Reinforcement Learning for Mixture-of-Expert Dialogue Management","date":"2023-02-21","arxiv_id":"2302.10850","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-block","title":"Reinforcement Learning for Block Decomposition of CAD Models","date":"2023-02-21","arxiv_id":"2302.11066","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-a-birth-and-death","title":"Reinforcement Learning in a Birth and Death Process: Breaking the Dependence on the State Space","date":"2023-02-21","arxiv_id":"2302.10667","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-path-planning-employing-mpc-reinforcement","title":"UAV Path Planning Employing MPC- Reinforcement Learning Method Considering Collision Avoidance","date":"2023-02-21","arxiv_id":"2302.10669","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-function","title":"Reinforcement Learning with Function Approximation: From Linear to Nonlinear","date":"2023-02-20","arxiv_id":"2302.09703","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-deep-reinforcement-learning-by-verifying","title":"Safe Deep Reinforcement Learning by Verifying Task-Level Properties","date":"2023-02-20","arxiv_id":"2302.10030","repositories_listed":0,"syntology":null},{"url":null,"slug":"autodoviz-human-centered-automation-for","title":"AutoDOViz: Human-Centered Automation for Decision Optimization","date":"2023-02-19","arxiv_id":"2302.09688","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositionality-and-bounds-for-optimal-value","title":"Compositionality and Bounds for Optimal Value Functions in Reinforcement Learning","date":"2023-02-19","arxiv_id":"2302.09676","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-video-corpus-moment-retrieval","title":"Interactive Video Corpus Moment Retrieval using Reinforcement Learning","date":"2023-02-19","arxiv_id":"2302.09522","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-and-versatile-bipedal-jumping-control","title":"Robust and Versatile Bipedal Jumping Control through Reinforcement Learning","date":"2023-02-19","arxiv_id":"2302.09450","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-multimodal-reinforcement-learning","title":"Effective Multimodal Reinforcement Learning with Modality Alignment and Importance Enhancement","date":"2023-02-18","arxiv_id":"2302.09318","repositories_listed":0,"syntology":null},{"url":null,"slug":"promoting-cooperation-in-multi-agent","title":"Promoting Cooperation in Multi-Agent Reinforcement Learning via Mutual Help","date":"2023-02-18","arxiv_id":"2302.09277","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-state-augmentation-based-approach-to","title":"A State Augmentation based approach to Reinforcement Learning from Human Preferences","date":"2023-02-17","arxiv_id":"2302.08734","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-reward-initialization-for","title":"Data Driven Reward Initialization for Preference based Reinforcement Learning","date":"2023-02-17","arxiv_id":"2302.08733","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-mmwave","title":"Deep Reinforcement Learning for mmWave Initial Beam Alignment","date":"2023-02-17","arxiv_id":"2302.08969","repositories_listed":0,"syntology":null}],"record_sha256":"fed63398c4896c87b05eebd4cd4fd33612e65deaa5d6a4f068b59e20721796d7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}