{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/the-nethack-learning-environment","title":"The NetHack Learning Environment","arxiv_id":"2006.13760","date":"2020-06-24","proceeding":"NeurIPS 2020 12","authors":["Heinrich Küttler","Nantas Nardelli","Alexander H. Miller","Roberta Raileanu","Marco Selvatici","Edward Grefenstette","Tim Rocktäschel"],"abstract":"Progress in Reinforcement Learning (RL) algorithms goes hand-in-hand with the development of challenging environments that test the limits of current methods. While existing RL environments are either sufficiently complex or based on fast simulation, they are rarely both. Here, we present the NetHack Learning Environment (NLE), a scalable, procedurally generated, stochastic, rich, and challenging environment for RL research based on the popular single-player terminal-based roguelike game, NetHack. We argue that NetHack is sufficiently complex to drive long-term research on problems such as exploration, planning, skill acquisition, and language-conditioned RL, while dramatically reducing the computational resources required to gather a large amount of experience. We compare NLE and its task suite to existing alternatives, and discuss why it is an ideal medium for testing the robustness and systematic generalization of RL agents. We demonstrate empirical success for early stages of the game using a distributed Deep RL baseline and Random Network Distillation exploration, alongside qualitative analysis of various agents trained in the environment. NLE is open source at https://github.com/facebookresearch/nle.","url_abs":"https://arxiv.org/abs/2006.13760v2","url_pdf":"https://arxiv.org/pdf/2006.13760v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"the-nethack-learning-environment","repo_url":"https://github.com/facebookresearch/nle","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"none","reach":null},{"paper_slug":"the-nethack-learning-environment","repo_url":"https://github.com/Pieter-Cawood/Reinforcement-Learning","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"the-nethack-learning-environment","repo_url":"https://github.com/Sarah-wookey/RL_NetHack_2020","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}}],"tasks":[{"task_slug":"nethack","task_name":"NetHack"},{"task_slug":"score","task_name":"NetHack Score"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"systematic-generalization","task_name":"Systematic Generalization"}],"methods":[],"datasets_introduced":[{"slug":"nethack-learning-environment","name":"NetHack Learning Environment","full_name":"NetHack Learning Environment"}],"methods_introduced":[],"results":[{"leaderboard":"/sota/score-on-nethack-learning-environment","task":"NetHack Score","dataset":"NetHack Learning Environment","model":"RND-mon-hum-neu-mal","rank_in_archive_order":1,"of":1,"metrics":{"Average Score":"780"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2006.13760","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.13760"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"deterministic:regex_extraction","url":"https://github.com/facebookresearch/nle","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Sarah-wookey/RL_NetHack_2020","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Pieter-Cawood/Reinforcement-Learning","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"ran_draft_wrong":1,"unverified":10},"by_repo_kind":{"official":{"samples":3,"ran":1,"repositories":1},"listed":{"samples":8,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":3,"samples":[{"code_sha256_prefix":"2fa2a5c4cf69b215","entry":"nested_map","repo":"facebookresearch/nle","repo_kind":"official","path":"nle/agent/agent.py","file_url":"https://github.com/facebookresearch/nle/blob/HEAD/nle/agent/agent.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"2fa2a5c4cf69b215"}},{"code_sha256_prefix":"e624e7bc82f85ecd","entry":"compute_baseline_loss","repo":"facebookresearch/nle","repo_kind":"official","path":"nle/agent/agent.py","file_url":"https://github.com/facebookresearch/nle/blob/HEAD/nle/agent/agent.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"e624e7bc82f85ecd"}},{"code_sha256_prefix":"3fba94f2f8bc0565","entry":"compute_entropy_loss","repo":"facebookresearch/nle","repo_kind":"official","path":"nle/agent/agent.py","file_url":"https://github.com/facebookresearch/nle/blob/HEAD/nle/agent/agent.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"3fba94f2f8bc0565"}},{"code_sha256_prefix":"5685e488bd317dd0","entry":"crop_glyphs","repo":"Pieter-Cawood/Reinforcement-Learning","repo_kind":"listed","path":"NLE_A2C/MyAgent2.py","file_url":"https://github.com/Pieter-Cawood/Reinforcement-Learning/blob/HEAD/NLE_A2C/MyAgent2.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"5685e488bd317dd0"}},{"code_sha256_prefix":"6b6543f92a642e1f","entry":"crop_state","repo":"Pieter-Cawood/Reinforcement-Learning","repo_kind":"listed","path":"NLE_A2C/MyAgent2.py","file_url":"https://github.com/Pieter-Cawood/Reinforcement-Learning/blob/HEAD/NLE_A2C/MyAgent2.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"6b6543f92a642e1f"}},{"code_sha256_prefix":"c89816ac8ce23ea2","entry":"crop_state","repo":"Pieter-Cawood/Reinforcement-Learning","repo_kind":"listed","path":"NLE_A2C/a2c_mod.py","file_url":"https://github.com/Pieter-Cawood/Reinforcement-Learning/blob/HEAD/NLE_A2C/a2c_mod.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"c89816ac8ce23ea2"}},{"code_sha256_prefix":"8f61913dfe74ccf7","entry":"get_views","repo":"Pieter-Cawood/Reinforcement-Learning","repo_kind":"listed","path":"NLE_MBS/Model_Train_File_Policies.py","file_url":"https://github.com/Pieter-Cawood/Reinforcement-Learning/blob/HEAD/NLE_MBS/Model_Train_File_Policies.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"8f61913dfe74ccf7"}},{"code_sha256_prefix":"2f6ab6f543c54d76","entry":"make_env","repo":"Pieter-Cawood/Reinforcement-Learning","repo_kind":"listed","path":"NLE_DQN/Agent.py","file_url":"https://github.com/Pieter-Cawood/Reinforcement-Learning/blob/HEAD/NLE_DQN/Agent.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"2f6ab6f543c54d76"}},{"code_sha256_prefix":"b9fc7ecce3b69c96","entry":"random_walk","repo":"Pieter-Cawood/Reinforcement-Learning","repo_kind":"listed","path":"NLE_MBS/Model_Train_File_Policies.py","file_url":"https://github.com/Pieter-Cawood/Reinforcement-Learning/blob/HEAD/NLE_MBS/Model_Train_File_Policies.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"b9fc7ecce3b69c96"}},{"code_sha256_prefix":"38d19bc77768e47c","entry":"run_episode","repo":"Pieter-Cawood/Reinforcement-Learning","repo_kind":"listed","path":"NLE_Method_Comparison/A2C_1/evaluation.py","file_url":"https://github.com/Pieter-Cawood/Reinforcement-Learning/blob/HEAD/NLE_Method_Comparison/A2C_1/evaluation.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"38d19bc77768e47c"}},{"code_sha256_prefix":"2cb4d2b63925bbc6","entry":"transform_observation","repo":"Pieter-Cawood/Reinforcement-Learning","repo_kind":"listed","path":"NLE_DQN/Agent.py","file_url":"https://github.com/Pieter-Cawood/Reinforcement-Learning/blob/HEAD/NLE_DQN/Agent.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"2cb4d2b63925bbc6"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}