{"url":"/task/montezumas-revenge","name":"Montezuma's Revenge","slug":"montezumas-revenge","description_markdown":"Montezuma's Revenge is an ATARI 2600 Benchmark game that is known to be difficult to perform on for reinforcement learning algorithms. Solutions typically employ algorithms that incentivise environment exploration in different ways.\r\n\r\nFor the state-of-the art tables, please consult the parent Atari Games task.\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [Q-map](https://github.com/fabiopardo/qmap) )</span>","categories":[{"name":"Playing Games","url":"/area/playing-games"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":61,"papers_with_code":31,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/montezuma-s-revenge-on-atari-2600-montezuma-s","slug":"montezuma-s-revenge-on-atari-2600-montezuma-s","dataset":"Atari 2600 Montezuma's Revenge","dataset_url":"/dataset/arcade-learning-environment","rows_in_archive":3,"metrics":["Average Return (NoOp)"],"first_row_in_archive_order":{"model":"Flare","paper_title":"Reinforcement Learning with Latent Flow","paper_url":"/paper/reinforcement-learning-with-latent-flow-1","paper_date":"2021-01-06","arxiv_id":"2101.01857","code_links":[{"title":"WendyShang/flare","url":"https://github.com/WendyShang/flare"},{"title":"WendyShang/dqn_zoo","url":"https://github.com/WendyShang/dqn_zoo"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/arcade-learning-environment","name":"Arcade Learning Environment","full_name":"Arcade Learning Environment","num_papers_in_archive":366}],"subtasks":[],"parent_tasks":[{"url":"/task/atari-games","name":"Atari Games"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":31,"tagged_in_all":61,"items":[{"url":"/paper/rainbow-combining-improvements-in-deep","title":"Rainbow: Combining Improvements in Deep Reinforcement Learning","date":"2017-10-06","arxiv_id":"1710.02298","repositories_listed":34,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/exploration-by-random-network-distillation","title":"Exploration by Random Network Distillation","date":"2018-10-30","arxiv_id":"1810.12894","repositories_listed":22,"syntology":{"n":43,"n_ran":26,"n_unverified":17,"n_pointer_only":15}},{"url":"/paper/hierarchical-deep-reinforcement-learning","title":"Hierarchical Deep Reinforcement Learning: Integrating Temporal Abstraction and Intrinsic Motivation","date":"2016-04-20","arxiv_id":"1604.06057","repositories_listed":4,"syntology":null},{"url":"/paper/go-explore-a-new-approach-for-hard","title":"Go-Explore: a New Approach for Hard-Exploration Problems","date":"2019-01-30","arxiv_id":"1901.10995","repositories_listed":3,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":4}},{"url":"/paper/a-study-of-global-and-episodic-bonuses-for","title":"A Study of Global and Episodic Bonuses for Exploration in Contextual MDPs","date":"2023-06-05","arxiv_id":"2306.03236","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-with-latent-flow-1","title":"Reinforcement Learning with Latent Flow","date":"2021-01-06","arxiv_id":"2101.01857","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/first-return-then-explore","title":"First return, then explore","date":"2020-04-27","arxiv_id":"2004.12919","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":4}},{"url":"/paper/exploring-unknown-states-with-action-balance","title":"Exploring Unknown States with Action Balance","date":"2020-03-10","arxiv_id":"2003.04518","repositories_listed":2,"syntology":null},{"url":"/paper/q-map-a-convolutional-approach-for-goal","title":"Scaling All-Goals Updates in Reinforcement Learning Using Convolutional Neural Networks","date":"2018-10-06","arxiv_id":"1810.02927","repositories_listed":2,"syntology":null},{"url":"/paper/2505-10819","title":"PoE-World: Compositional World Modeling with Products of Programmatic Experts","date":"2025-05-16","arxiv_id":"2505.10819","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-of-plasticity-loss-in-on-policy-deep","title":"A Study of Plasticity Loss in On-Policy Deep Reinforcement Learning","date":"2024-05-29","arxiv_id":"2405.19153","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_unverified":4,"n_pointer_only":11}},{"url":"/paper/fine-tuning-reinforcement-learning-models-is","title":"Fine-tuning Reinforcement Learning Models is Secretly a Forgetting Mitigation Problem","date":"2024-02-05","arxiv_id":"2402.02868","repositories_listed":1,"syntology":null},{"url":"/paper/flipping-coins-to-estimate-pseudocounts-for","title":"Flipping Coins to Estimate Pseudocounts for Exploration in Reinforcement Learning","date":"2023-06-05","arxiv_id":"2306.03186","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/redeeming-intrinsic-rewards-via-constrained","title":"Redeeming Intrinsic Rewards via Constrained Optimization","date":"2022-11-14","arxiv_id":"2211.07627","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/hybrid-rl-using-both-offline-and-online-data","title":"Hybrid RL: Using Both Offline and Online Data Can Make RL Efficient","date":"2022-10-13","arxiv_id":"2210.06718","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/cell-free-latent-go-explore","title":"Cell-Free Latent Go-Explore","date":"2022-08-31","arxiv_id":"2208.14928","repositories_listed":1,"syntology":null},{"url":"/paper/open-ended-reinforcement-learning-with-neural","title":"Open-Ended Reinforcement Learning with Neural Reward Functions","date":"2022-02-16","arxiv_id":"2202.08266","repositories_listed":1,"syntology":null},{"url":"/paper/noveld-a-simple-yet-effective-exploration","title":"NovelD: A Simple yet Effective Exploration Criterion","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-abstract-models-for-strategic","title":"Learning Abstract Models for Strategic Exploration and Fast Reward Transfer","date":"2020-07-12","arxiv_id":"2007.05896","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-sensitive-learning-and-planning-1","title":"Uncertainty-sensitive Learning and Planning with Ensembles","date":"2019-12-19","arxiv_id":"1912.09996","repositories_listed":1,"syntology":null},{"url":"/paper/deepsynth-program-synthesis-for-automatic","title":"DeepSynth: Automata Synthesis for Automatic Task Segmentation in Deep Reinforcement Learning","date":"2019-11-22","arxiv_id":"1911.10244","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-sensitive-learning-and-planning","title":"Uncertainty - sensitive learning and planning with ensembles","date":"2019-09-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/combining-experience-replay-with-exploration","title":"Combining Experience Replay with Exploration by Random Network Distillation","date":"2019-05-18","arxiv_id":"1905.07579","repositories_listed":1,"syntology":null},{"url":"/paper/using-natural-language-for-reward-shaping-in","title":"Using Natural Language for Reward Shaping in Reinforcement Learning","date":"2019-03-05","arxiv_id":"1903.02020","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/empowerment-driven-exploration-using-mutual","title":"Empowerment-driven Exploration using Mutual Information Estimation","date":"2018-10-11","arxiv_id":"1810.05533","repositories_listed":1,"syntology":null},{"url":"/paper/playing-hard-exploration-games-by-watching","title":"Playing hard exploration games by watching YouTube","date":"2018-05-29","arxiv_id":"1805.11592","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/feature-control-as-intrinsic-motivation-for","title":"Feature Control as Intrinsic Motivation for Hierarchical Reinforcement Learning","date":"2017-05-18","arxiv_id":"1705.06769","repositories_listed":1,"syntology":null},{"url":"/paper/beating-atari-with-natural-language-guided","title":"Beating Atari with Natural Language Guided Reinforcement Learning","date":"2017-04-18","arxiv_id":"1704.05539","repositories_listed":1,"syntology":null},{"url":"/paper/count-based-exploration-with-neural-density","title":"Count-Based Exploration with Neural Density Models","date":"2017-03-03","arxiv_id":"1703.01310","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/playing-atari-games-with-deep-reinforcement","title":"Playing Atari Games with Deep Reinforcement Learning and Human Checkpoint Replay","date":"2016-07-18","arxiv_id":"1607.05077","repositories_listed":1,"syntology":null}],"syntology_records":12,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}