{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/hierarchical-reinforcement-learning/papers/2","list_of":"/task/hierarchical-reinforcement-learning","task":"Hierarchical Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":384,"counts":{"archive_papers_tagged":384,"with_a_code_link":111,"where_syntology_ran_a_sample":22,"not_listed_spam_title":0,"listed":384,"listed_where_code_ran":22,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":22,"every_run_a_failure_of_syntologys_instrument":0,"listed_with_a_run_with_no_instrument_failure":22,"listed_every_run_a_failure_of_syntologys_instrument":0,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/hierarchical-reinforcement-learning","prev":"/task/hierarchical-reinforcement-learning","next":"/task/hierarchical-reinforcement-learning/papers/3","papers":[{"url":"/paper/learning-actionable-representations-with-goal","slug":"learning-actionable-representations-with-goal","title":"Learning Actionable Representations with Goal-Conditioned Policies","date":"2018-11-19","arxiv_id":"1811.07819","repositories_listed":1,"syntology":null},{"url":"/paper/diversity-driven-extensible-hierarchical","slug":"diversity-driven-extensible-hierarchical","title":"Diversity-Driven Extensible Hierarchical Reinforcement Learning","date":"2018-11-10","arxiv_id":"1811.04324","repositories_listed":1,"syntology":null},{"url":"/paper/keep-it-stupid-simple","slug":"keep-it-stupid-simple","title":"Combining imagination and heuristics to learn strategies that generalize","date":"2018-09-10","arxiv_id":"1809.03406","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-semantic-parsing-for-if-then","slug":"interactive-semantic-parsing-for-if-then","title":"Interactive Semantic Parsing for If-Then Recipes via Hierarchical Reinforcement Learning","date":"2018-08-21","arxiv_id":"1808.06740","repositories_listed":1,"syntology":null},{"url":"/paper/safe-option-critic-learning-safety-in-the","slug":"safe-option-critic-learning-safety-in-the","title":"Safe Option-Critic: Learning Safety in the Option-Critic Architecture","date":"2018-07-21","arxiv_id":"1807.08060","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-for-zero","slug":"hierarchical-reinforcement-learning-for-zero","title":"Hierarchical Reinforcement Learning for Zero-shot Generalization with Subtask Dependencies","date":"2018-07-19","arxiv_id":"1807.07665","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/hierarchical-reinforcement-learning-for-zero#ran","syntology_url":"https://syntology.ai/paper/1807.07665","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.07665"}},"official":{"repos":["srsohn/subtask-graph-execution"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/using-reward-machines-for-high-level-task","slug":"using-reward-machines-for-high-level-task","title":"Using Reward Machines for High-Level Task Specification and Decomposition in Reinforcement Learning","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/logically-constrained-reinforcement-learning","slug":"logically-constrained-reinforcement-learning","title":"Logically-Constrained Reinforcement Learning","date":"2018-01-24","arxiv_id":"1801.08099","repositories_listed":1,"syntology":null},{"url":"/paper/crossmodal-attentive-skill-learner","slug":"crossmodal-attentive-skill-learner","title":"Crossmodal Attentive Skill Learner","date":"2017-11-28","arxiv_id":"1711.10314","repositories_listed":1,"syntology":null},{"url":"/paper/feature-control-as-intrinsic-motivation-for","slug":"feature-control-as-intrinsic-motivation-for","title":"Feature Control as Intrinsic Motivation for Hierarchical Reinforcement Learning","date":"2017-05-18","arxiv_id":"1705.06769","repositories_listed":1,"syntology":null},{"url":"/paper/feudal-networks-for-hierarchical","slug":"feudal-networks-for-hierarchical","title":"FeUdal Networks for Hierarchical Reinforcement Learning","date":"2017-03-03","arxiv_id":"1703.01161","repositories_listed":1,"syntology":null},{"url":null,"slug":"strict-subgoal-execution-reliable-long","title":"Strict Subgoal Execution: Reliable Long-Horizon Planning in Hierarchical Reinforcement Learning","date":"2025-06-26","arxiv_id":"2506.21039","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-and-value","title":"Hierarchical Reinforcement Learning and Value Optimization for Challenging Quadruped Locomotion","date":"2025-06-24","arxiv_id":"2506.20036","repositories_listed":0,"syntology":null},{"url":null,"slug":"tailored-conversations-beyond-llms-a-rl-based","title":"Tailored Conversations beyond LLMs: A RL-Based Dialogue Manager","date":"2025-06-24","arxiv_id":"2506.19652","repositories_listed":0,"syntology":null},{"url":null,"slug":"hilight-a-hierarchical-reinforcement-learning","title":"HiLight: A Hierarchical Reinforcement Learning Framework with Global Adversarial Guidance for Large-Scale Traffic Signal Control","date":"2025-06-17","arxiv_id":"2506.14391","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-temporal-structure-an-overview-of","title":"Discovering Temporal Structure: An Overview of Hierarchical Reinforcement Learning","date":"2025-06-16","arxiv_id":"2506.14045","repositories_listed":0,"syntology":null},{"url":null,"slug":"discounting-and-drug-seeking-in-biological","title":"Discounting and Drug Seeking in Biological Hierarchical Reinforcement Learning","date":"2025-06-05","arxiv_id":"2506.04549","repositories_listed":0,"syntology":null},{"url":null,"slug":"mrsd-multi-resolution-skill-discovery-for-hrl","title":"MRSD: Multi-Resolution Skill Discovery for HRL Agents","date":"2025-05-27","arxiv_id":"2505.21410","repositories_listed":0,"syntology":null},{"url":null,"slug":"let-humanoids-hike-integrative-skill","title":"Let Humanoids Hike! Integrative Skill Development on Complex Trails","date":"2025-05-09","arxiv_id":"2505.06218","repositories_listed":0,"syntology":null},{"url":null,"slug":"mastering-multi-drone-volleyball-through","title":"Mastering Multi-Drone Volleyball through Hierarchical Co-Self-Play Reinforcement Learning","date":"2025-05-07","arxiv_id":"2505.04317","repositories_listed":0,"syntology":null},{"url":null,"slug":"reem-ensemble-building-thermodynamics-model","title":"ReeM: Ensemble Building Thermodynamics Model for Efficient HVAC Control via Hierarchical Reinforcement Learning","date":"2025-05-05","arxiv_id":"2505.02439","repositories_listed":0,"syntology":null},{"url":null,"slug":"d3hrl-a-distributed-hierarchical","title":"D3HRL: A Distributed Hierarchical Reinforcement Learning Approach Based on Causal Discovery and Spurious Correlation Detection","date":"2025-05-04","arxiv_id":"2505.01979","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-in-multi","title":"Hierarchical Reinforcement Learning in Multi-Goal Spatial Navigation with Autonomous Mobile Robots","date":"2025-04-26","arxiv_id":"2504.18794","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-hierarchical-reinforcement-learning","title":"Federated Hierarchical Reinforcement Learning for Adaptive Traffic Signal Control","date":"2025-04-07","arxiv_id":"2504.05553","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-tale-of-two-goals-leveraging-sequentiality","title":"A tale of two goals: leveraging sequentiality in multi-goal scenarios","date":"2025-03-27","arxiv_id":"2503.21677","repositories_listed":0,"syntology":null},{"url":null,"slug":"option-discovery-using-llm-guided-semantic","title":"Option Discovery Using LLM-guided Semantic Hierarchical Reinforcement Learning","date":"2025-03-24","arxiv_id":"2503.19007","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-safe","title":"Hierarchical Reinforcement Learning for Safe Mapless Navigation with Congestion Estimation","date":"2025-03-15","arxiv_id":"2503.12036","repositories_listed":0,"syntology":null},{"url":null,"slug":"dhp-discrete-hierarchical-planning-for","title":"DHP: Discrete Hierarchical Planning for Hierarchical Reinforcement Learning Agents","date":"2025-02-04","arxiv_id":"2502.01956","repositories_listed":0,"syntology":null},{"url":null,"slug":"certificated-actor-critic-hierarchical","title":"Certificated Actor-Critic: Hierarchical Reinforcement Learning with Control Barrier Functions for Safe Navigation","date":"2025-01-29","arxiv_id":"2501.17424","repositories_listed":0,"syntology":null},{"url":null,"slug":"extensive-exploration-in-complex-traffic","title":"Extensive Exploration in Complex Traffic Scenarios using Hierarchical Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.14992","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-reinforcement-learning","title":"A Hierarchical Reinforcement Learning Framework for Multi-UAV Combat Using Leader-Follower Strategy","date":"2025-01-22","arxiv_id":"2501.13132","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-driven-hierarchical-reinforcement","title":"Attention-Driven Hierarchical Reinforcement Learning with Particle Filtering for Source Localization in Dynamic Fields","date":"2025-01-22","arxiv_id":"2501.13084","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-11","title":"Hierarchical Reinforcement Learning for Optimal Agent Grouping in Cooperative Systems","date":"2025-01-11","arxiv_id":"2501.06554","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-workplace-productivity-and-well","title":"Enhancing Workplace Productivity and Well-being Using AI Agent","date":"2025-01-04","arxiv_id":"2501.02368","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-hierarchical-reinforcement-learning","title":"Scalable Hierarchical Reinforcement Learning for Hyper Scale Multi-Robot Task Planning","date":"2024-12-27","arxiv_id":"2412.19538","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-geospatial-search-for-efficient-tenant","title":"Active Geospatial Search for Efficient Tenant Eviction Outreach","date":"2024-12-19","arxiv_id":"2412.17854","repositories_listed":0,"syntology":null},{"url":null,"slug":"simulation-free-hierarchical-latent-policy","title":"Simulation-Free Hierarchical Latent Policy Planning for Proactive Dialogues","date":"2024-12-19","arxiv_id":"2412.14584","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-task-generalisation-with-multi","title":"Accelerating Task Generalisation with Multi-Level Skill Hierarchies","date":"2024-11-05","arxiv_id":"2411.02998","repositories_listed":0,"syntology":null},{"url":null,"slug":"guiding-multi-agent-multi-task-reinforcement","title":"Guiding Multi-agent Multi-task Reinforcement Learning by a Hierarchical Framework with Logical Reward Shaping","date":"2024-11-02","arxiv_id":"2411.01184","repositories_listed":0,"syntology":null},{"url":null,"slug":"demystifying-linear-mdps-and-novel-dynamics","title":"Demystifying Linear MDPs and Novel Dynamics Aggregation Framework","date":"2024-10-31","arxiv_id":"2410.24089","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualpredicator-learning-abstract-world","title":"VisualPredicator: Learning Abstract World Models with Neuro-Symbolic Predicates for Robot Planning","date":"2024-10-30","arxiv_id":"2410.23156","repositories_listed":0,"syntology":null},{"url":null,"slug":"copyright-aware-incentive-scheme-for","title":"Copyright-Aware Incentive Scheme for Generative Art Models Using Hierarchical Reinforcement Learning","date":"2024-10-26","arxiv_id":"2410.20180","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforced-trader-hrt-a-bi-level","title":"Hierarchical Reinforced Trader (HRT): A Bi-Level Approach for Optimizing Stock Selection and Execution","date":"2024-10-19","arxiv_id":"2410.14927","repositories_listed":0,"syntology":null},{"url":null,"slug":"recoverychaining-learning-local-recovery","title":"RecoveryChaining: Learning Local Recovery Policies for Robust Manipulation","date":"2024-10-17","arxiv_id":"2410.13979","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-unsupervised-skill-discovery-for","title":"Disentangled Unsupervised Skill Discovery for Efficient Hierarchical Reinforcement Learning","date":"2024-10-15","arxiv_id":"2410.11251","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-universal-value-function","title":"Hierarchical Universal Value Function Approximators","date":"2024-10-11","arxiv_id":"2410.08997","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-hierarchical-reinforcement-learning","title":"Offline Hierarchical Reinforcement Learning via Inverse Optimization","date":"2024-10-10","arxiv_id":"2410.07933","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-skill-discovery-for-robotic","title":"Unsupervised Skill Discovery for Robotic Manipulation through Automatic Task Generation","date":"2024-10-07","arxiv_id":"2410.04855","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-meets-options-hierarchical","title":"Diffusion Meets Options: Hierarchical Generative Skill Composition for Temporally-Extended Tasks","date":"2024-10-03","arxiv_id":"2410.02389","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-10","title":"Hierarchical Reinforcement Learning for Temporal Abstraction of Listwise Recommendation","date":"2024-09-11","arxiv_id":"2409.07416","repositories_listed":0,"syntology":null},{"url":null,"slug":"mastering-the-digital-art-of-war-developing","title":"Mastering the Digital Art of War: Developing Intelligent Combat Simulation Agents for Wargaming Using Hierarchical Reinforcement Learning","date":"2024-08-23","arxiv_id":"2408.13333","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-in-context-reinforcement","title":"Retrieval-Augmented Hierarchical in-Context Reinforcement Learning and Hindsight Modular Reflections for Task Planning with LLMs","date":"2024-08-12","arxiv_id":"2408.06520","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-abstraction-in-reinforcement-1","title":"Temporal Abstraction in Reinforcement Learning with Offline Data","date":"2024-07-21","arxiv_id":"2407.15241","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-average-reward-linearly-solvable","title":"Hierarchical Average-Reward Linearly-solvable Markov Decision Processes","date":"2024-07-09","arxiv_id":"2407.06690","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-provably-efficient-option-based-algorithm","title":"A Provably Efficient Option-Based Algorithm for both High-Level and Low-Level Learning","date":"2024-06-21","arxiv_id":"2406.15124","repositories_listed":0,"syntology":null},{"url":null,"slug":"dipper-direct-preference-optimization-to","title":"DIPPER: Direct Preference Optimization to Accelerate Primitive-Enabled Hierarchical Reinforcement Learning","date":"2024-06-16","arxiv_id":"2406.10892","repositories_listed":0,"syntology":null},{"url":null,"slug":"racon-retrieval-augmented-simulated-character","title":"RACon: Retrieval-Augmented Simulated Character Locomotion Control","date":"2024-06-11","arxiv_id":"2406.17795","repositories_listed":0,"syntology":null},{"url":null,"slug":"lgr2-language-guided-reward-relabeling-for","title":"LGR2: Language Guided Reward Relabeling for Accelerating Hierarchical Reinforcement Learning","date":"2024-06-09","arxiv_id":"2406.05881","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-words-to-actions-unveiling-the","title":"From Words to Actions: Unveiling the Theoretical Underpinnings of LLM-Driven Autonomous Systems","date":"2024-05-30","arxiv_id":"2405.19883","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-of-machine-learning-enabled-1","title":"An Overview of Machine Learning-Enabled Optimization for Reconfigurable Intelligent Surfaces-Aided 6G Networks: From Reinforcement Learning to Large Language Models","date":"2024-05-09","arxiv_id":"2405.17439","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-optimization-on-uplink-ofdma-and-mu","title":"Joint Optimization on Uplink OFDMA and MU-MIMO for IEEE 802.11ax: Deep Hierarchical Reinforcement Learning Approach","date":"2024-04-03","arxiv_id":"2404.02486","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-options","title":"Reinforcement Learning with Options and State Representation","date":"2024-03-16","arxiv_id":"2403.10855","repositories_listed":0,"syntology":null},{"url":null,"slug":"smaug-a-sliding-multidimensional-task-window","title":"SMAUG: A Sliding Multidimensional Task Window-Based MARL Framework for Adaptive Real-Time Subtask Recognition","date":"2024-03-04","arxiv_id":"2403.01816","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-session-based-recommendation-via","title":"Explainable Session-based Recommendation via Path Reasoning","date":"2024-02-28","arxiv_id":"2403.00832","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-artificial-intelligence-for-digital","title":"Scaling Artificial Intelligence for Digital Wargaming in Support of Decision-Making","date":"2024-02-08","arxiv_id":"2402.06075","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-intelligent-agents-in-combat","title":"Scaling Intelligent Agents in Combat Simulations for Wargaming","date":"2024-02-08","arxiv_id":"2402.06694","repositories_listed":0,"syntology":null},{"url":null,"slug":"toponav-topological-navigation-for-efficient","title":"TopoNav: Topological Navigation for Efficient Exploration in Sparse Reward Environments","date":"2024-02-06","arxiv_id":"2402.04061","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergency-computing-an-adaptive-collaborative","title":"Emergency Computing: An Adaptive Collaborative Inference Method Based on Hierarchical Reinforcement Learning","date":"2024-02-03","arxiv_id":"2402.02146","repositories_listed":0,"syntology":null},{"url":null,"slug":"slim-skill-learning-with-multiple-critics","title":"SLIM: Skill Learning with Multiple Critics","date":"2024-02-01","arxiv_id":"2402.00823","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-temporal-interplay-in-human-mobility","title":"Spatial-Temporal Interplay in Human Mobility: A Hierarchical Reinforcement Learning Approach with Hypergraph Representation","date":"2023-12-25","arxiv_id":"2312.15717","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-session-budget-optimization-for-forward","title":"Multi-Session Budget Optimization for Forward Auction-based Federated Learning","date":"2023-11-21","arxiv_id":"2311.12548","repositories_listed":0,"syntology":null},{"url":null,"slug":"imagination-augmented-hierarchical","title":"Imagination-Augmented Hierarchical Reinforcement Learning for Safe and Interactive Autonomous Driving in Urban Environments","date":"2023-11-17","arxiv_id":"2311.10309","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-neuro-inspired-hierarchical-reinforcement","title":"A Central Motor System Inspired Pre-training Reinforcement Learning for Robotic Control","date":"2023-11-14","arxiv_id":"2311.07822","repositories_listed":0,"syntology":null},{"url":null,"slug":"mtac-hierarchical-reinforcement-learning","title":"MTAC: Hierarchical Reinforcement Learning-based Multi-gait Terrain-adaptive Quadruped Controller","date":"2023-11-01","arxiv_id":"2401.03337","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-decision-transformer-via","title":"Rethinking Decision Transformer via Hierarchical Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00267","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-extrinsic-dexterity-with","title":"Learning Extrinsic Dexterity with Parameterized Manipulation Primitives","date":"2023-10-26","arxiv_id":"2310.17785","repositories_listed":0,"syntology":null},{"url":null,"slug":"forecaster-towards-temporally-abstract-tree","title":"Forecaster: Towards Temporally Abstract Tree-Search Planning from Pixels","date":"2023-10-16","arxiv_id":"2310.09997","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-9","title":"Hierarchical Reinforcement Learning for Temporal Pattern Prediction","date":"2023-10-09","arxiv_id":"2310.05695","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-hierarchical-reinforcement-learning-for","title":"Safe Hierarchical Reinforcement Learning for CubeSat Task Scheduling Based on Energy Consumption","date":"2023-09-21","arxiv_id":"2309.12004","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-with-6","title":"Hierarchical reinforcement learning with natural language subgoals","date":"2023-09-20","arxiv_id":"2309.11564","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-space-abstraction-in-hierarchical-1","title":"Goal Space Abstraction in Hierarchical Reinforcement Learning via Set-Based Reachability Analysis","date":"2023-09-14","arxiv_id":"2309.07675","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-space-abstraction-in-hierarchical","title":"Goal Space Abstraction in Hierarchical Reinforcement Learning via Reachability Analysis","date":"2023-09-12","arxiv_id":"2309.07168","repositories_listed":0,"syntology":null},{"url":null,"slug":"spread-control-method-on-unknown-networks","title":"Spread Control Method on Unknown Networks Based on Hierarchical Reinforcement Learning","date":"2023-08-28","arxiv_id":"2308.14311","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-band-assignment-and-beam-management","title":"Joint Band Assignment and Beam Management using Hierarchical Reinforcement Learning for Multi-Band Communication","date":"2023-08-25","arxiv_id":"2308.13202","repositories_listed":0,"syntology":null},{"url":null,"slug":"wasserstein-diversity-enriched-regularizer","title":"Wasserstein Diversity-Enriched Regularizer for Hierarchical Reinforcement Learning","date":"2023-08-02","arxiv_id":"2308.00989","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-efficient-orchestrations-for","title":"Communication-Efficient Orchestrations for URLLC Service via Hierarchical Reinforcement Learning","date":"2023-07-25","arxiv_id":"2307.13415","repositories_listed":0,"syntology":null},{"url":null,"slug":"vehicle-dispatching-and-routing-of-on-demand","title":"Vehicle Dispatching and Routing of On-Demand Intercity Ride-Pooling Services: A Multi-Agent Hierarchical Reinforcement Learning Approach","date":"2023-07-13","arxiv_id":"2307.06742","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-hierarchical-interactive-multi","title":"Learning Hierarchical Interactive Multi-Object Search for Mobile Manipulation","date":"2023-07-12","arxiv_id":"2307.06125","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-empowerment-towards-tractable","title":"Hierarchical Empowerment: Towards Tractable Empowerment-Based Skill Learning","date":"2023-07-06","arxiv_id":"2307.02728","repositories_listed":0,"syntology":null},{"url":null,"slug":"landmark-guided-active-exploration-with","title":"Landmark Guided Active Exploration with State-specific Balance Coefficient","date":"2023-06-30","arxiv_id":"2306.17484","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-and-trajectory-planning-for-urban","title":"Action and Trajectory Planning for Urban Autonomous Driving with Hierarchical Reinforcement Learning","date":"2023-06-28","arxiv_id":"2306.15968","repositories_listed":0,"syntology":null},{"url":null,"slug":"int-hrl-towards-intention-based-hierarchical","title":"Int-HRL: Towards Intention-based Hierarchical Reinforcement Learning","date":"2023-06-20","arxiv_id":"2306.11483","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-deduction-path-learning-via","title":"Automatic Deduction Path Learning via Reinforcement Learning with Environmental Correction","date":"2023-06-16","arxiv_id":"2306.10083","repositories_listed":0,"syntology":null},{"url":null,"slug":"skill-critic-refining-learned-skills-for","title":"Skill-Critic: Refining Learned Skills for Hierarchical Reinforcement Learning","date":"2023-06-14","arxiv_id":"2306.08388","repositories_listed":0,"syntology":null},{"url":null,"slug":"pear-primitive-enabled-adaptive-relabeling","title":"PEAR: Primitive enabled Adaptive Relabeling for boosting Hierarchical Reinforcement Learning","date":"2023-06-10","arxiv_id":"2306.06394","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-sparse-conversations-for-improved","title":"CAVEN: An Embodied Conversational Agent for Efficient Audio-Visual Navigation in Noisy Environments","date":"2023-06-06","arxiv_id":"2306.04047","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-8","title":"Hierarchical Reinforcement Learning for Modeling User Novelty-Seeking Intent in Recommender Systems","date":"2023-06-02","arxiv_id":"2306.01476","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-option-dependent-analysis-of-regret","title":"An Option-Dependent Analysis of Regret Minimization Algorithms in Finite-Horizon Semi-Markov Decision Processes","date":"2023-05-10","arxiv_id":"2305.06936","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-learn-group-alignment-a-self","title":"Learning to Learn Group Alignment: A Self-Tuning Credo Framework with Multiagent Teams","date":"2023-04-14","arxiv_id":"2304.07337","repositories_listed":0,"syntology":null},{"url":null,"slug":"crisp-curriculum-inducing-primitive-informed","title":"CRISP: Curriculum Inducing Primitive Informed Subgoal Prediction for Hierarchical Reinforcement Learning","date":"2023-04-07","arxiv_id":"2304.03535","repositories_listed":0,"syntology":null}],"record_sha256":"53ee3f5c1ef2f671dc575e8ad38efc4f5d5d4c550bdecb32008325c6711d04ca","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}