{"url":"/task/nethack","name":"NetHack","slug":"nethack","description_markdown":"Mean in-game score over 1000 episodes with random seeds not seen during training. See https://arxiv.org/abs/2006.13760 (Section 2.4 Evaluation Protocol) for details.","categories":[{"name":"Playing Games","url":"/area/playing-games"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":28,"papers_with_code":22,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":1,"parent_tasks":0},"benchmarks":[],"datasets":[],"subtasks":[{"url":"/task/score","name":"NetHack Score"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":22,"of":22,"tagged_in_all":28,"items":[{"url":"/paper/the-nethack-learning-environment","title":"The NetHack Learning Environment","date":"2020-06-24","arxiv_id":"2006.13760","repositories_listed":3,"syntology":{"n":11,"n_ran":1,"n_unverified":10,"n_pointer_only":3}},{"url":"/paper/hierarchical-kickstarting-for-skill-transfer","title":"Hierarchical Kickstarting for Skill Transfer in Reinforcement Learning","date":"2022-07-23","arxiv_id":"2207.11584","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/cora-benchmarks-baselines-and-metrics-as-a","title":"CORA: Benchmarks, Baselines, and Metrics as a Platform for Continual Reinforcement Learning Agents","date":"2021-10-19","arxiv_id":"2110.10067","repositories_listed":2,"syntology":null},{"url":"/paper/bebold-exploration-beyond-the-boundary-of-1","title":"BeBold: Exploration Beyond the Boundary of Explored Regions","date":"2020-12-15","arxiv_id":"2012.08621","repositories_listed":2,"syntology":null},{"url":"/paper/syllabus-portable-curricula-for-reinforcement","title":"Syllabus: Portable Curricula for Reinforcement Learning Agents","date":"2024-11-18","arxiv_id":"2411.11318","repositories_listed":1,"syntology":null},{"url":"/paper/online-intrinsic-rewards-for-decision-making","title":"Online Intrinsic Rewards for Decision Making Agents from Large Language Model Feedback","date":"2024-10-30","arxiv_id":"2410.23022","repositories_listed":1,"syntology":{"n":12,"n_ran":0,"n_unverified":12,"n_pointer_only":12}},{"url":"/paper/pufferlib-making-reinforcement-learning","title":"PufferLib: Making Reinforcement Learning Libraries and Environments Play Nice","date":"2024-06-11","arxiv_id":"2406.12905","repositories_listed":1,"syntology":null},{"url":"/paper/playing-nethack-with-llms-potential","title":"Playing NetHack with LLMs: Potential & Limitations as Zero-Shot Agents","date":"2024-03-01","arxiv_id":"2403.00690","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/craftax-a-lightning-fast-benchmark-for-open","title":"Craftax: A Lightning-Fast Benchmark for Open-Ended Reinforcement Learning","date":"2024-02-26","arxiv_id":"2402.16801","repositories_listed":1,"syntology":{"n":16,"n_ran":2,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/skill-set-optimization-reinforcing-language","title":"Skill Set Optimization: Reinforcing Language Model Behavior via Transferable Skills","date":"2024-02-05","arxiv_id":"2402.03244","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/fine-tuning-reinforcement-learning-models-is","title":"Fine-tuning Reinforcement Learning Models is Secretly a Forgetting Mitigation Problem","date":"2024-02-05","arxiv_id":"2402.02868","repositories_listed":1,"syntology":null},{"url":"/paper/diff-history-for-long-context-language-agents","title":"diff History for Neural Language Agents","date":"2023-12-12","arxiv_id":"2312.07540","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/motif-intrinsic-motivation-from-artificial","title":"Motif: Intrinsic Motivation from Artificial Intelligence Feedback","date":"2023-09-29","arxiv_id":"2310.00166","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":5}},{"url":"/paper/scaling-laws-for-imitation-learning-in","title":"Scaling Laws for Imitation Learning in Single-Agent Games","date":"2023-07-18","arxiv_id":"2307.09423","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/luckymera-a-modular-ai-framework-for-building","title":"LuckyMera: a Modular AI Framework for Building Hybrid NetHack Agents","date":"2023-07-17","arxiv_id":"2307.08532","repositories_listed":1,"syntology":null},{"url":"/paper/katakomba-tools-and-benchmarks-for-data","title":"Katakomba: Tools and Benchmarks for Data-Driven NetHack","date":"2023-06-14","arxiv_id":"2306.08772","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/dungeons-and-data-a-large-scale-nethack","title":"Dungeons and Data: A Large-Scale NetHack Dataset","date":"2022-11-01","arxiv_id":"2211.00539","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/improving-policy-learning-via-language","title":"Improving Policy Learning via Language Dynamics Distillation","date":"2022-09-30","arxiv_id":"2210.00066","repositories_listed":1,"syntology":{"n":8,"n_ran":2,"n_unverified":6,"n_pointer_only":4}},{"url":"/paper/insights-from-the-neurips-2021-nethack","title":"Insights From the NeurIPS 2021 NetHack Challenge","date":"2022-03-22","arxiv_id":"2203.11889","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/noveld-a-simple-yet-effective-exploration","title":"NovelD: A Simple yet Effective Exploration Criterion","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/silg-the-multi-environment-symbolic","title":"SILG: The Multi-environment Symbolic Interactive Language Grounding Benchmark","date":"2021-10-20","arxiv_id":"2110.10661","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/minihack-the-planet-a-sandbox-for-open-ended","title":"MiniHack the Planet: A Sandbox for Open-Ended Reinforcement Learning Research","date":"2021-09-27","arxiv_id":"2109.13202","repositories_listed":1,"syntology":null}],"syntology_records":14,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}