{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/the-option-critic-architecture","title":"The Option-Critic Architecture","arxiv_id":"1609.05140","date":"2016-09-16","proceeding":null,"authors":["Pierre-Luc Bacon","Jean Harb","Doina Precup"],"abstract":"Temporal abstraction is key to scaling up learning and planning in\nreinforcement learning. While planning with temporally extended actions is well\nunderstood, creating such abstractions autonomously from data has remained\nchallenging. We tackle this problem in the framework of options [Sutton, Precup\n& Singh, 1999; Precup, 2000]. We derive policy gradient theorems for options\nand propose a new option-critic architecture capable of learning both the\ninternal policies and the termination conditions of options, in tandem with the\npolicy over options, and without the need to provide any additional rewards or\nsubgoals. Experimental results in both discrete and continuous environments\nshowcase the flexibility and efficiency of the framework.","url_abs":"http://arxiv.org/abs/1609.05140v2","url_pdf":"http://arxiv.org/pdf/1609.05140v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"the-option-critic-architecture","repo_url":"https://github.com/AshrithSagar/option-critic","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"the-option-critic-architecture","repo_url":"https://github.com/UWaterloo-ASL/option-critic-architecture","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"the-option-critic-architecture","repo_url":"https://github.com/abdulhaim/moe_options","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"the-option-critic-architecture","repo_url":"https://github.com/abdulhaim/options_pytorch_image_env","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"the-option-critic-architecture","repo_url":"https://github.com/arushi12130/SafeOptionCritic","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"the-option-critic-architecture","repo_url":"https://github.com/elitalobo/Hierarchical-RL-Algorithms","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"the-option-critic-architecture","repo_url":"https://github.com/huiwenzhang/oc-tf","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}},{"paper_slug":"the-option-critic-architecture","repo_url":"https://github.com/jeanharb/option_critic","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"the-option-critic-architecture","repo_url":"https://github.com/lweitkamp/option-critic-pytorch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"the-option-critic-architecture","repo_url":"https://github.com/yadrimz/option-critic","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}}],"tasks":[{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1609.05140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1609.05140"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/huiwenzhang/oc-tf","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/jeanharb/option_critic","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/abdulhaim/moe_options","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/lweitkamp/option-critic-pytorch","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/AshrithSagar/option-critic","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/elitalobo/Hierarchical-RL-Algorithms","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/arushi12130/SafeOptionCritic","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/yadrimz/option-critic","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/UWaterloo-ASL/option-critic-architecture","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/abdulhaim/options_pytorch_image_env","reach":{"status":"ok"}}],"summary":{"ran_honours":1,"ran_fixture":1,"ran_draft_wrong":1,"unverified":3},"by_repo_kind":{"listed":{"samples":6,"ran":3,"repositories":2}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":3,"samples":[{"code_sha256_prefix":"7ebafde922686c8a","entry":"cov","repo":"elitalobo/Hierarchical-RL-Algorithms","repo_kind":"listed","path":"option-critic/hierarchical_dqn.py","file_url":"https://github.com/elitalobo/Hierarchical-RL-Algorithms/blob/HEAD/option-critic/hierarchical_dqn.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"7ebafde922686c8a"}},{"code_sha256_prefix":"31775f7dff0837c8","entry":"cov","repo":"elitalobo/Hierarchical-RL-Algorithms","repo_kind":"listed","path":"option-critic/hierarchical_dqn.py","file_url":"https://github.com/elitalobo/Hierarchical-RL-Algorithms/blob/HEAD/option-critic/hierarchical_dqn.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"31775f7dff0837c8"}},{"code_sha256_prefix":"e4490f9789c3a31e","entry":"softmax","repo":"elitalobo/Hierarchical-RL-Algorithms","repo_kind":"listed","path":"option-critic/hierarchical_dqn.py","file_url":"https://github.com/elitalobo/Hierarchical-RL-Algorithms/blob/HEAD/option-critic/hierarchical_dqn.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"e4490f9789c3a31e"}},{"code_sha256_prefix":"dcb0930c599f1313","entry":"get_agent_meth","repo":"AshrithSagar/option-critic","repo_kind":"listed","path":"oca/agents/utils.py","file_url":"https://github.com/AshrithSagar/option-critic/blob/HEAD/oca/agents/utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"dcb0930c599f1313"}},{"code_sha256_prefix":"e9c61c409e6c07d6","entry":"make_env","repo":"AshrithSagar/option-critic","repo_kind":"listed","path":"oca/envs/utils.py","file_url":"https://github.com/AshrithSagar/option-critic/blob/HEAD/oca/envs/utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"e9c61c409e6c07d6"}},{"code_sha256_prefix":"9488a34878d9dd7e","entry":"to_tensor","repo":"AshrithSagar/option-critic","repo_kind":"listed","path":"oca/envs/utils.py","file_url":"https://github.com/AshrithSagar/option-critic/blob/HEAD/oca/envs/utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"9488a34878d9dd7e"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}