{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/hierarchical-reinforcement-learning/papers/4","list_of":"/task/hierarchical-reinforcement-learning","task":"Hierarchical Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":4,"rows_per_page":100,"rows":[301,384],"of":384,"counts":{"archive_papers_tagged":384,"with_a_code_link":111,"where_syntology_ran_a_sample":22,"not_listed_spam_title":0,"listed":384,"listed_where_code_ran":22,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":22,"every_run_a_failure_of_syntologys_instrument":0,"listed_with_a_run_with_no_instrument_failure":22,"listed_every_run_a_failure_of_syntologys_instrument":0,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/hierarchical-reinforcement-learning","prev":"/task/hierarchical-reinforcement-learning/papers/3","next":null,"papers":[{"url":null,"slug":"discovering-motor-programs-by-recomposing","title":"Discovering Motor Programs by Recomposing Demonstrations","date":"2020-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-with-abstract-learned-models-while","title":"Planning with Abstract Learned Models While Learning Transferable Subtasks","date":"2019-12-16","arxiv_id":"1912.07544","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-exploration-through-intrinsic","title":"Efficient Exploration through Intrinsic Motivation Learning for Unsupervised Subgoal Discovery in Model-Free Hierarchical Reinforcement Learning","date":"2019-11-18","arxiv_id":"1911.10164","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-method","title":"Hierarchical Reinforcement Learning Method for Autonomous Vehicle Behavior Planning","date":"2019-11-09","arxiv_id":"1911.03799","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-like-decision-making-document-level","title":"Human-Like Decision Making: Document-level Aspect Sentiment Classification via Hierarchical Reinforcement Learning","date":"2019-10-21","arxiv_id":"1910.09260","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-hierarchical-reinforcement","title":"Multi-agent Hierarchical Reinforcement Learning with Dynamic Termination","date":"2019-10-21","arxiv_id":"1910.09508","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-meta-reinforcement-learning-via-1","title":"MGHRL: Meta Goal-generation for Hierarchical Reinforcement Learning","date":"2019-09-30","arxiv_id":"1909.13607","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-atari-ball-games-with-hierarchical","title":"Playing Atari Ball Games with Hierarchical Reinforcement Learning","date":"2019-09-27","arxiv_id":"1909.12465","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-functionally-decomposed-hierarchies","title":"Learning Functionally Decomposed Hierarchies for Continuous Navigation Tasks","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-temporal-abstraction-with","title":"Learning Temporal Abstraction with Information-theoretic Constraints for Hierarchical Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-hierarchical-reinforcement-1","title":"Multi-Agent Hierarchical Reinforcement Learning for Humanoid Navigation","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-benefits-of-deep-hierarchical-rl","title":"PROVABLY BENEFITS OF DEEP HIERARCHICAL RL","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-skill-coding-learning-behavioral","title":"Sparse Skill Coding: Learning Behavioral Hierarchies with Sparse Codes","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"why-does-hierarchy-sometimes-work-so-well-in","title":"Why Does Hierarchy (Sometimes) Work So Well in Reinforcement Learning?","date":"2019-09-23","arxiv_id":"1909.10618","repositories_listed":0,"syntology":null},{"url":null,"slug":"ledeepchef-deep-reinforcement-learning-agent","title":"LeDeepChef: Deep Reinforcement Learning Agent for Families of Text-Based Games","date":"2019-09-04","arxiv_id":"1909.01646","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-sit-synthesizing-human-chair","title":"Learning to Sit: Synthesizing Human-Chair Interactions via Hierarchical Control","date":"2019-08-20","arxiv_id":"1908.07423","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-discovery-of-decision-states-for","title":"IR-VIC: Unsupervised Discovery of Sub-goals for Transfer in RL","date":"2019-07-24","arxiv_id":"1907.10580","repositories_listed":0,"syntology":null},{"url":null,"slug":"composing-diverse-policies-for-temporally","title":"Composing Diverse Policies for Temporally Extended Tasks","date":"2019-07-18","arxiv_id":"1907.08199","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-high-level-planning-symbols-from","title":"Learning High-Level Planning Symbols from Intrinsically Motivated Experience","date":"2019-07-18","arxiv_id":"1907.08313","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-world-graphs-to-accelerate","title":"Learning World Graphs to Accelerate Hierarchical Reinforcement Learning","date":"2019-07-01","arxiv_id":"1907.00664","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularized-hierarchical-policies-for","title":"Compositional Transfer in Hierarchical Reinforcement Learning","date":"2019-06-26","arxiv_id":"1906.11228","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-competitive","title":"Reinforcement Learning with Competitive Ensembles of Information-Constrained Primitives","date":"2019-06-25","arxiv_id":"1906.10667","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-networks-with-motivation","title":"Neural networks with motivation","date":"2019-06-23","arxiv_id":"1906.09528","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-skill-embeddings-for","title":"Disentangled Skill Embeddings for Reinforcement Learning","date":"2019-06-21","arxiv_id":"1906.09223","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-policy-adaptation-for-hierarchical","title":"Sub-policy Adaptation for Hierarchical Reinforcement Learning","date":"2019-06-13","arxiv_id":"1906.05862","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for","title":"Hierarchical Reinforcement Learning for Quadruped Locomotion","date":"2019-05-22","arxiv_id":"1905.08926","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-policy-adaptation-for-hierarchical-1","title":"Sub-policy Adaptation for Hierarchical Reinforcement Learning","date":"2019-05-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-and-exploiting-multiple-subgoals-for","title":"Learning and Exploiting Multiple Subgoals for Fast Exploration in Hierarchical Reinforcement Learning","date":"2019-05-13","arxiv_id":"1905.05180","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-actionable-representations-with-goal-1","title":"Learning Actionable Representations with Goal Conditioned Policies","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-control-visual-abstractions-for","title":"Learning to Control Visual Abstractions for Structured Exploration in Deep Reinforcement Learning","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dot-to-dot-achieving-structured-robotic","title":"Dot-to-Dot: Explainable Hierarchical Reinforcement Learning for Robotic Manipulation","date":"2019-04-14","arxiv_id":"1904.06703","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-task-discovery-with-limited-supervision-a","title":"Sub-Task Discovery with Limited Supervision: A Constrained Clustering Approach","date":"2019-03-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-hierarchical-reinforcement-learning-1","title":"Deep Hierarchical Reinforcement Learning Based Recommendations via Multi-goals Abstraction","date":"2019-03-22","arxiv_id":"1903.09374","repositories_listed":0,"syntology":null},{"url":null,"slug":"rloc-neurobiologically-inspired-hierarchical","title":"RLOC: Neurobiologically Inspired Hierarchical Reinforcement Learning Algorithm for Continuous Control of Nonlinear Dynamical Systems","date":"2019-03-07","arxiv_id":"1903.03064","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-in-hierarchical-reinforcement","title":"Planning in Hierarchical Reinforcement Learning: Guarantees for Using Local Policies","date":"2019-02-26","arxiv_id":"1902.10140","repositories_listed":0,"syntology":null},{"url":null,"slug":"aggregating-e-commerce-search-results-from","title":"Aggregating E-commerce Search Results from Heterogeneous Sources via Hierarchical Reinforcement Learning","date":"2019-02-24","arxiv_id":"1902.08882","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-coagent-networks-stochastic","title":"Asynchronous Coagent Networks","date":"2019-02-15","arxiv_id":"1902.05650","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-multi","title":"Hierarchical Reinforcement Learning for Multi-agent MOBA Game","date":"2019-01-23","arxiv_id":"1901.08004","repositories_listed":0,"syntology":null},{"url":null,"slug":"escape-room-a-configurable-testbed-for","title":"Escape Room: A Configurable Testbed for Hierarchical Reinforcement Learning","date":"2018-12-22","arxiv_id":"1812.09521","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperbolic-embeddings-for-learning-options-in","title":"Hyperbolic Embeddings for Learning Options in Hierarchical Reinforcement Learning","date":"2018-12-04","arxiv_id":"1812.01487","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-option-critic","title":"Natural Option Critic","date":"2018-12-04","arxiv_id":"1812.01488","repositories_listed":0,"syntology":null},{"url":null,"slug":"relation-mention-extraction-from-noisy-data","title":"Relation Mention Extraction from Noisy Data with Hierarchical Reinforcement Learning","date":"2018-11-03","arxiv_id":"1811.01237","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-representations-in-model-free","title":"Learning Representations in Model-Free Hierarchical Reinforcement Learning","date":"2018-10-23","arxiv_id":"1810.10096","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-sub-domain-modeling-for-dialogue","title":"Autonomous Sub-domain Modeling for Dialogue Policy with Hierarchical Deep Reinforcement Learning","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-hierarchical-reinforcement","title":"Incremental Hierarchical Reinforcement Learning with Multitask LMDPs","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"successor-options-an-option-discovery-1","title":"Successor Options : An Option Discovery Algorithm for Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-arithmetic-expression-calculator","title":"Neural Arithmetic Expression Calculator","date":"2018-09-23","arxiv_id":"1809.08590","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-reinforcement-learning-for-full-length","title":"On Reinforcement Learning for Full-length Game of StarCraft","date":"2018-09-23","arxiv_id":"1809.09095","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-interrupt-a-hierarchical-deep","title":"Learning to Interrupt: A Hierarchical Deep Reinforcement Learning Framework for Efficient Exploration","date":"2018-07-30","arxiv_id":"1807.11150","repositories_listed":0,"syntology":null},{"url":null,"slug":"representational-efficiency-outweighs-action","title":"Representational efficiency outweighs action efficiency in human program induction","date":"2018-07-18","arxiv_id":"1807.07134","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-nlp","title":"Deep Reinforcement Learning for NLP","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-with","title":"Hierarchical Reinforcement Learning with Abductive Planning","date":"2018-06-28","arxiv_id":"1806.10792","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-formation-of-the-structure-of","title":"Automatic formation of the structure of abstract machines in hierarchical reinforcement learning with state clustering","date":"2018-06-13","arxiv_id":"1806.05292","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-consistent-trajectory-autoencoder","title":"Self-Consistent Trajectory Autoencoder: Hierarchical Reinforcement Learning with Trajectory Embeddings","date":"2018-06-07","arxiv_id":"1806.02813","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-with-1","title":"Hierarchical Reinforcement Learning with Hindsight","date":"2018-05-21","arxiv_id":"1805.08180","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-with-deep","title":"Hierarchical Reinforcement Learning with Deep Nested Agents","date":"2018-05-18","arxiv_id":"1805.07008","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-hierarchical-reinforcement-learning","title":"Deep Hierarchical Reinforcement Learning Algorithm in Partially Observable Markov Decision Processes","date":"2018-05-11","arxiv_id":"1805.04419","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-hierarchical-reinforcement","title":"Multimodal Hierarchical Reinforcement Learning Policy for Task-Oriented Visual Dialog","date":"2018-05-08","arxiv_id":"1805.03257","repositories_listed":0,"syntology":null},{"url":null,"slug":"peorl-integrating-symbolic-planning-and","title":"PEORL: Integrating Symbolic Planning and Hierarchical Reinforcement Learning for Robust Decision-Making","date":"2018-04-20","arxiv_id":"1804.07779","repositories_listed":0,"syntology":null},{"url":null,"slug":"subgoal-discovery-for-hierarchical-dialogue","title":"Subgoal Discovery for Hierarchical Dialogue Policy Learning","date":"2018-04-20","arxiv_id":"1804.07855","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-space-policies-for-hierarchical","title":"Latent Space Policies for Hierarchical Reinforcement Learning","date":"2018-04-09","arxiv_id":"1804.02808","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning","title":"Hierarchical Reinforcement Learning: Approximating Optimal Discounted TSP Using Local Policies","date":"2018-03-13","arxiv_id":"1803.04674","repositories_listed":0,"syntology":null},{"url":null,"slug":"automata-guided-hierarchical-reinforcement-1","title":"AUTOMATA GUIDED HIERARCHICAL REINFORCEMENT LEARNING FOR ZERO-SHOT SKILL COMPOSITION","date":"2018-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-agent-for-disentangling","title":"Universal Agent for Disentangling Environments and Tasks","date":"2018-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-eigenoption-critic-framework","title":"The Eigenoption-Critic Framework","date":"2017-12-11","arxiv_id":"1712.04065","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-captioning-via-hierarchical","title":"Video Captioning via Hierarchical Reinforcement Learning","date":"2017-11-29","arxiv_id":"1711.11135","repositories_listed":0,"syntology":null},{"url":null,"slug":"automata-guided-hierarchical-reinforcement","title":"Automata-Guided Hierarchical Reinforcement Learning for Skill Composition","date":"2017-10-31","arxiv_id":"1711.00129","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-subtask-discovery-with-non","title":"Hierarchical Subtask Discovery With Non-Negative Matrix Factorization","date":"2017-08-01","arxiv_id":"1708.00463","repositories_listed":0,"syntology":null},{"url":null,"slug":"plan-attend-generate-character-level-neural","title":"Plan, Attend, Generate: Character-Level Neural Machine Translation with Planning","date":"2017-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-domain-modelling-for-dialogue-management","title":"Sub-domain Modelling for Dialogue Management with Hierarchical Reinforcement Learning","date":"2017-06-19","arxiv_id":"1706.06210","repositories_listed":0,"syntology":null},{"url":null,"slug":"situational-awareness-by-risk-conscious","title":"Situational Awareness by Risk-Conscious Skills","date":"2016-10-10","arxiv_id":"1610.02847","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-stage-temporal-difference-learning-for","title":"Multi-Stage Temporal Difference Learning for 2048-like Games","date":"2016-06-23","arxiv_id":"1606.07374","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-reinforcement-learning-method","title":"A Hierarchical Reinforcement Learning Method for Persistent Time-Sensitive Tasks","date":"2016-06-20","arxiv_id":"1606.06355","repositories_listed":0,"syntology":null},{"url":null,"slug":"option-discovery-in-hierarchical","title":"Option Discovery in Hierarchical Reinforcement Learning using Spatio-Temporal Clustering","date":"2016-05-17","arxiv_id":"1605.05359","repositories_listed":0,"syntology":null},{"url":null,"slug":"classifying-options-for-deep-reinforcement","title":"Classifying Options for Deep Reinforcement Learning","date":"2016-04-27","arxiv_id":"1604.08153","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithms-for-batch-hierarchical","title":"Algorithms for Batch Hierarchical Reinforcement Learning","date":"2016-03-29","arxiv_id":"1603.08869","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-linearly-solvable-markov","title":"Hierarchical Linearly-Solvable Markov Decision Problems","date":"2016-03-10","arxiv_id":"1603.03267","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-constrained-and-adaptive","title":"A Framework for Constrained and Adaptive Behavior-Based Agents","date":"2015-06-07","arxiv_id":"1506.02312","repositories_listed":0,"syntology":null},{"url":null,"slug":"grounding-hierarchical-reinforcement-learning","title":"Grounding Hierarchical Reinforcement Learning Models for Knowledge Transfer","date":"2014-12-19","arxiv_id":"1412.6451","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-principles-of-the-hippocampal","title":"Design Principles of the Hippocampal Cognitive Map","date":"2014-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-hierarchical-reinforcement-learning","title":"Bayesian Hierarchical Reinforcement Learning","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-dialogue-policy-learning-using","title":"Hierarchical Dialogue Policy Learning using Flexible State Transitions and Linear Function Approximation","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"optimising-incremental-dialogue-decisions","title":"Optimising Incremental Dialogue Decisions Using Information Density for Interactive Systems","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-hmms-and-bayesian-networks-for","title":"Comparing HMMs and Bayesian Networks for Surface Realisation","date":"2012-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"195f69d2c1f8b897baef4df8f3a2c46435a9b2b532e3c3107906f5926b65ea18","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}