{"url":"/dataset/arcade-learning-environment","name":"Arcade Learning Environment","full_name":"Arcade Learning Environment","description_markdown":"The **Arcade Learning Environment** (ALE) is an object-oriented framework that allows researchers to develop AI agents for Atari 2600 games. It is built on top of the Atari 2600 emulator Stella and separates the details of emulation from agent design.\r\n\r\nSource: [https://github.com/mgbellemare/Arcade-Learning-Environment](https://github.com/mgbellemare/Arcade-Learning-Environment)\r\nImage Source: [https://github.com/muupan/async-rl/blob/master/README.md](https://github.com/muupan/async-rl/blob/master/README.md)","description_withheld":null,"homepage":"https://github.com/mgbellemare/Arcade-Learning-Environment","introduced_date":"2012-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/the-arcade-learning-environment-an-evaluation","title":"The Arcade Learning Environment: An Evaluation Platform for General Agents","first_author":"Marc G. Bellemare","url":null},"license":{"name":"GPL-2","url":"https://github.com/mgbellemare/Arcade-Learning-Environment/blob/master/LICENSE.md"},"modalities":[{"name":"Environment","url":"/datasets/modality/environment"}],"tasks":[{"name":"Atari Games","url":"/task/atari-games","datasets_with_task":"/datasets/task/atari-games"},{"name":"Montezuma's Revenge","url":"/task/montezumas-revenge","datasets_with_task":"/datasets/task/montezumas-revenge"}],"languages":[],"variants":["Atari 2600 Breakout","Arcade Learning Environment","Atari 2600 Yars Revenge","Atari 2600 Wizard of Wor","Atari 2600 Video Pinball","Atari 2600 Up and Down","Atari 2600 Tutankham","Atari 2600 Time Pilot","Atari 2600 Surround","Atari 2600 Space Invaders","Atari 2600 Seaquest","Atari 2600 Robotank","Atari 2600 Road Runner","Atari 2600 River Raid","Atari 2600 Private Eye","Atari 2600 Pitfall!","Atari 2600 Name This Game","Atari 2600 Ms. Pacman","Atari 2600 Montezuma's Revenge","Atari 2600 Kung-Fu Master","Atari 2600 Kangaroo","Atari 2600 Journey Escape","Atari 2600 James Bond","Atari 2600 Ice Hockey","Atari 2600 Gravitar","Atari 2600 Frostbite","Atari 2600 Freeway","Atari 2600 Fishing Derby","Atari 2600 Elevator Action","Atari 2600 Double Dunk","Atari 2600 Demon Attack","Atari 2600 Crazy Climber","Atari 2600 Chopper Command","Atari 2600 Centipede","Atari 2600 Carnival","Atari 2600 Beam Rider","Atari 2600 Battle Zone","Atari 2600 Bank Heist","Atari 2600 Atlantis","Atari 2600 Asteroids","Atari 2600 Assault","Atari-57","Atari 2600 Zaxxon","Atari 2600 Venture","Atari 2600 Tennis","Atari 2600 Skiing","Atari 2600 Q*Bert","Atari 2600 Pooyan","Atari 2600 Pong","Atari 2600 Phoenix","Atari 2600 Krull","Atari 2600 HERO","Atari 2600 Gopher","Atari 2600 Bowling","Atari 2600 Berzerk","Atari 2600 Amidar","Atari 2600 Alien"],"data_loaders":[{"repo":"https://github.com/mgbellemare/Arcade-Learning-Environment","url":"https://github.com/mgbellemare/Arcade-Learning-Environment","frameworks":[]}],"num_papers_in_archive":366,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/atari-games-on-atari-2600-freeway","task":"Atari Games","dataset_variant":"Atari 2600 Freeway","rows":59,"metrics":["Score"],"first_row_in_archive_order":{"model":"TRPO-hash","paper":"/paper/exploration-a-study-of-count-based","metrics":{"Score":"34.0"},"code_links":[{"title":"uoe-agents/derl","url":"https://github.com/uoe-agents/derl"},{"title":"nhynes/abc","url":"https://github.com/nhynes/abc"},{"title":"clementbernardd/Count-Based-Exploration","url":"https://github.com/clementbernardd/Count-Based-Exploration"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-breakout","task":"Atari Games","dataset_variant":"Atari 2600 Breakout","rows":58,"metrics":["Score"],"first_row_in_archive_order":{"model":"GDI-H3(200M frames)","paper":"/paper/generalized-data-distribution-iteration","metrics":{"Score":"864.00"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-qbert","task":"Atari Games","dataset_variant":"Atari 2600 Q*Bert","rows":57,"metrics":["Score","Best Score","Return"],"first_row_in_archive_order":{"model":"Agent57","paper":"/paper/agent57-outperforming-the-atari-human","metrics":{"Score":"580328.14"},"code_links":[{"title":"michaelnny/deep_rl_zoo","url":"https://github.com/michaelnny/deep_rl_zoo"},{"title":"pocokhc/agent57","url":"https://github.com/pocokhc/agent57"},{"title":"yuta0821/agent57_pytorch","url":"https://github.com/yuta0821/agent57_pytorch"},{"title":"Nkluge-correa/teeny-tiny_castle","url":"https://github.com/Nkluge-correa/teeny-tiny_castle"},{"title":"YHL04/agent57","url":"https://github.com/YHL04/agent57"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-seaquest","task":"Atari Games","dataset_variant":"Atari 2600 Seaquest","rows":57,"metrics":["Score","Return"],"first_row_in_archive_order":{"model":"GDI-H3(200M frames)","paper":"/paper/generalized-data-distribution-iteration","metrics":{"Score":"1000000"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-space-invaders","task":"Atari Games","dataset_variant":"Atari 2600 Space Invaders","rows":55,"metrics":["Score","Best Score","Return"],"first_row_in_archive_order":{"model":"GDI-H3(200M frames)","paper":"/paper/generalized-data-distribution-iteration","metrics":{"Score":"154380"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-venture","task":"Atari Games","dataset_variant":"Atari 2600 Venture","rows":55,"metrics":["Score"],"first_row_in_archive_order":{"model":"Agent57","paper":"/paper/agent57-outperforming-the-atari-human","metrics":{"Score":"2623.71"},"code_links":[{"title":"michaelnny/deep_rl_zoo","url":"https://github.com/michaelnny/deep_rl_zoo"},{"title":"pocokhc/agent57","url":"https://github.com/pocokhc/agent57"},{"title":"yuta0821/agent57_pytorch","url":"https://github.com/yuta0821/agent57_pytorch"},{"title":"Nkluge-correa/teeny-tiny_castle","url":"https://github.com/Nkluge-correa/teeny-tiny_castle"},{"title":"YHL04/agent57","url":"https://github.com/YHL04/agent57"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-frostbite","task":"Atari Games","dataset_variant":"Atari 2600 Frostbite","rows":53,"metrics":["Score","Best Score"],"first_row_in_archive_order":{"model":"MuZero","paper":"/paper/mastering-atari-go-chess-and-shogi-by","metrics":{"Score":"631378.53"},"code_links":[{"title":"werner-duvaud/muzero-general","url":"https://github.com/werner-duvaud/muzero-general"},{"title":"opendilab/LightZero","url":"https://github.com/opendilab/LightZero"},{"title":"koulanurag/muzero-pytorch","url":"https://github.com/koulanurag/muzero-pytorch"},{"title":"johan-gras/MuZero","url":"https://github.com/johan-gras/MuZero"},{"title":"kaesve/muzero","url":"https://github.com/kaesve/muzero"},{"title":"Zeta36/muzero","url":"https://github.com/Zeta36/muzero"},{"title":"YuriCat/MuZeroJupyterExample","url":"https://github.com/YuriCat/MuZeroJupyterExample"},{"title":"DHDev0/Muzero","url":"https://github.com/DHDev0/Muzero"},{"title":"foersterrobert/MuZero","url":"https://github.com/foersterrobert/MuZero"},{"title":"k-lombard/CS4641_Project","url":"https://github.com/k-lombard/CS4641_Project"},{"title":"k-lombard/Deep-Learning-Chess-AI","url":"https://github.com/k-lombard/Deep-Learning-Chess-AI"},{"title":"JuanCCS/muzero-jc","url":"https://github.com/JuanCCS/muzero-jc"},{"title":"ZiyuanMa/reversi","url":"https://github.com/ZiyuanMa/reversi"},{"title":"Miatto-research-group/muzero","url":"https://github.com/Miatto-research-group/muzero"},{"title":"colindbrown/columbia-deep-learning-project","url":"https://github.com/colindbrown/columbia-deep-learning-project"},{"title":"SHRIVP/muzero","url":"https://github.com/SHRIVP/muzero"},{"title":"dmiracle/muzero-starter","url":"https://github.com/dmiracle/muzero-starter"},{"title":"snjstudent/MyMuzero","url":"https://github.com/snjstudent/MyMuzero"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-gravitar","task":"Atari Games","dataset_variant":"Atari 2600 Gravitar","rows":53,"metrics":["Score"],"first_row_in_archive_order":{"model":"Agent57","paper":"/paper/agent57-outperforming-the-atari-human","metrics":{"Score":"19213.96"},"code_links":[{"title":"michaelnny/deep_rl_zoo","url":"https://github.com/michaelnny/deep_rl_zoo"},{"title":"pocokhc/agent57","url":"https://github.com/pocokhc/agent57"},{"title":"yuta0821/agent57_pytorch","url":"https://github.com/yuta0821/agent57_pytorch"},{"title":"Nkluge-correa/teeny-tiny_castle","url":"https://github.com/Nkluge-correa/teeny-tiny_castle"},{"title":"YHL04/agent57","url":"https://github.com/YHL04/agent57"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-pong","task":"Atari Games","dataset_variant":"Atari 2600 Pong","rows":52,"metrics":["Score"],"first_row_in_archive_order":{"model":"Duel noop","paper":"/paper/dueling-network-architectures-for-deep","metrics":{"Score":"21.0"},"code_links":[{"title":"labmlai/annotated_deep_learning_paper_implementations","url":"https://github.com/labmlai/annotated_deep_learning_paper_implementations"},{"title":"tensorpack/tensorpack","url":"https://github.com/tensorpack/tensorpack/tree/master/examples/DeepQNetwork"},{"title":"facebookresearch/Horizon","url":"https://github.com/facebookresearch/Horizon"},{"title":"facebookresearch/ReAgent","url":"https://github.com/facebookresearch/ReAgent"},{"title":"opendilab/DI-engine","url":"https://github.com/opendilab/DI-engine"},{"title":"NervanaSystems/coach","url":"https://github.com/NervanaSystems/coach"},{"title":"Curt-Park/rainbow-is-all-you-need","url":"https://github.com/Curt-Park/rainbow-is-all-you-need"},{"title":"chainer/chainerrl","url":"https://github.com/chainer/chainerrl"},{"title":"tensorlayer/RLzoo","url":"https://github.com/tensorlayer/RLzoo"},{"title":"marload/DeepRL-TensorFlow2","url":"https://github.com/marload/DeepRL-TensorFlow2"},{"title":"dxyang/DQN_pytorch","url":"https://github.com/dxyang/DQN_pytorch"},{"title":"philtabor/Deep-Q-Learning-Paper-To-Code","url":"https://github.com/philtabor/Deep-Q-Learning-Paper-To-Code"},{"title":"ku2482/sac-discrete.pytorch","url":"https://github.com/ku2482/sac-discrete.pytorch"},{"title":"cove9988/TradingGym","url":"https://github.com/cove9988/TradingGym"},{"title":"BY571/DQN-Atari-Agents","url":"https://github.com/BY571/DQN-Atari-Agents"},{"title":"atavakol/action-branching-agents","url":"https://github.com/atavakol/action-branching-agents"},{"title":"chandar-lab/RLHive","url":"https://github.com/chandar-lab/RLHive"},{"title":"JuliaPOMDP/DeepQLearning.jl","url":"https://github.com/JuliaPOMDP/DeepQLearning.jl"},{"title":"gouxiangchen/dueling-DQN-pytorch","url":"https://github.com/gouxiangchen/dueling-DQN-pytorch"},{"title":"R-Sweke/DeepQ-Decoding","url":"https://github.com/R-Sweke/DeepQ-Decoding"},{"title":"cocolico14/N-step-Dueling-DDQN-PER-Pacman","url":"https://github.com/cocolico14/N-step-Dueling-DDQN-PER-Pacman"},{"title":"Montherapy/Deep-reinforcement-learning-for-multi-class-imbalanced-classification","url":"https://github.com/Montherapy/Deep-reinforcement-learning-for-multi-class-imbalanced-classification"},{"title":"mindspore-courses/Deep-Reinforcement-Learning-Algorithms-with-MindSpore","url":"https://github.com/mindspore-courses/Deep-Reinforcement-Learning-Algorithms-with-MindSpore"},{"title":"kmdanielduan/DQN_Family_PyTorch","url":"https://github.com/kmdanielduan/DQN_Family_PyTorch"},{"title":"ethanmclark1/carla_aebs","url":"https://github.com/ethanmclark1/carla_aebs"},{"title":"clarky104/carla_aebs","url":"https://github.com/clarky104/carla_aebs"},{"title":"KDL-umass/saliency_maps","url":"https://github.com/KDL-umass/saliency_maps"},{"title":"hemilpanchiwala/Dueling-Network-Architectures","url":"https://github.com/hemilpanchiwala/Dueling-Network-Architectures"},{"title":"hemilpanchiwala/Dueling_Network_Architectures","url":"https://github.com/hemilpanchiwala/Dueling_Network_Architectures"},{"title":"utarumo/RL_implementation","url":"https://github.com/utarumo/RL_implementation"},{"title":"alessandrositta/Flatland_challenge","url":"https://github.com/alessandrositta/Flatland_challenge"},{"title":"MohammadAsadolahi/Dueling_Deep-Double_Q-Learning-for-solving-OpenAi-Gym-LunarLander-v2-in-python","url":"https://github.com/MohammadAsadolahi/Dueling_Deep-Double_Q-Learning-for-solving-OpenAi-Gym-LunarLander-v2-in-python"},{"title":"wtingda/DeepRLBreakout","url":"https://github.com/wtingda/DeepRLBreakout"},{"title":"mindspore-courses/Rainbow-MindSpore","url":"https://github.com/mindspore-courses/Rainbow-MindSpore"},{"title":"rybread1/deep-rl-trex","url":"https://github.com/rybread1/deep-rl-trex"},{"title":"rybread1/DeepRlTrex","url":"https://github.com/rybread1/DeepRlTrex"},{"title":"OMS1996/Carla_The_RL_Self-Driving-Car","url":"https://github.com/OMS1996/Carla_The_RL_Self-Driving-Car"},{"title":"xusophia/DataSciFinalProj","url":"https://github.com/xusophia/DataSciFinalProj"},{"title":"SayhoKim/tetrisRL","url":"https://github.com/SayhoKim/tetrisRL"},{"title":"nathanin/pad","url":"https://github.com/nathanin/pad"},{"title":"zynk13/dueling-dqn-Reinforcement-learning","url":"https://github.com/zynk13/dueling-dqn-Reinforcement-learning"},{"title":"near32/regym","url":"https://github.com/near32/regym"},{"title":"1jsingh/rl_navigation","url":"https://github.com/1jsingh/rl_navigation"},{"title":"prajwalgatti/DRL-Continuous-Control","url":"https://github.com/prajwalgatti/DRL-Continuous-Control"},{"title":"la3lma/chezjulia","url":"https://github.com/la3lma/chezjulia"},{"title":"la3lma/Chez","url":"https://github.com/la3lma/Chez"},{"title":"fengsterooni/dql","url":"https://github.com/fengsterooni/dql"},{"title":"Adrelf/DRL-navigation","url":"https://github.com/Adrelf/DRL-navigation"},{"title":"prajwalgatti/DRL-Navigation","url":"https://github.com/prajwalgatti/DRL-Navigation"},{"title":"MEOWMEOW114/nd893-p1-navigation-banana","url":"https://github.com/MEOWMEOW114/nd893-p1-navigation-banana"},{"title":"kshitij-ingale/Reinforcement-Learning","url":"https://github.com/kshitij-ingale/Reinforcement-Learning"},{"title":"botforge/simplementation","url":"https://github.com/botforge/simplementation"},{"title":"manvibharat/Stock-price-pridiction-using-Deep-reienforcement-learning","url":"https://github.com/manvibharat/Stock-price-pridiction-using-Deep-reienforcement-learning"},{"title":"HussonnoisMaxence/RL_Algorithms","url":"https://github.com/HussonnoisMaxence/RL_Algorithms"},{"title":"shehrum/RL_Navigation","url":"https://github.com/shehrum/RL_Navigation"},{"title":"mohit8935/Deep-Q-Learning-Paper","url":"https://github.com/mohit8935/Deep-Q-Learning-Paper"},{"title":"iDataist/Navigation-with-Deep-Q-Network","url":"https://github.com/iDataist/Navigation-with-Deep-Q-Network"},{"title":"jsztompka/DuelDQN","url":"https://github.com/jsztompka/DuelDQN"},{"title":"170928/-Review-Dueling-Deep-Q-Network","url":"https://github.com/170928/-Review-Dueling-Deep-Q-Network"},{"title":"Brandon-Rozek/DeepRL","url":"https://github.com/Brandon-Rozek/DeepRL"},{"title":"abryeemessi/Wednesday","url":"https://github.com/abryeemessi/Wednesday"},{"title":"eddynelson/dqn","url":"https://github.com/eddynelson/dqn"},{"title":"shashwatsaxena571/DRL-navigation","url":"https://github.com/shashwatsaxena571/DRL-navigation"},{"title":"JBGUIMBAUD/deep-reenforcement-learning","url":"https://github.com/JBGUIMBAUD/deep-reenforcement-learning"},{"title":"nbopardi/smb","url":"https://github.com/nbopardi/smb"},{"title":"ZainRaza14/deepRL","url":"https://github.com/ZainRaza14/deepRL"},{"title":"guillaumeboniface/bananaland","url":"https://github.com/guillaumeboniface/bananaland"},{"title":"MOVzeroOne/DQN","url":"https://github.com/MOVzeroOne/DQN"},{"title":"austinsilveria/Banana-Collection-DQN","url":"https://github.com/austinsilveria/Banana-Collection-DQN"},{"title":"mightypirate1/DRL-Tetris","url":"https://github.com/mightypirate1/DRL-Tetris"},{"title":"FaboNo/DRLND","url":"https://github.com/FaboNo/DRLND"},{"title":"opplieam/Pong-Deep-RL","url":"https://github.com/opplieam/Pong-Deep-RL"},{"title":"jezzarax/drlnd_p1_navigation","url":"https://github.com/jezzarax/drlnd_p1_navigation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-private-eye","task":"Atari Games","dataset_variant":"Atari 2600 Private Eye","rows":52,"metrics":["Score"],"first_row_in_archive_order":{"model":"Go-Explore","paper":"/paper/first-return-then-explore","metrics":{"Score":"95756"},"code_links":[{"title":"uber-research/go-explore","url":"https://github.com/uber-research/go-explore"},{"title":"qgallouedec/lge","url":"https://github.com/qgallouedec/lge"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-montezumas-revenge","task":"Atari Games","dataset_variant":"Atari 2600 Montezuma's Revenge","rows":50,"metrics":["Score"],"first_row_in_archive_order":{"model":"Go-Explore","paper":"/paper/first-return-then-explore","metrics":{"Score":"43791"},"code_links":[{"title":"uber-research/go-explore","url":"https://github.com/uber-research/go-explore"},{"title":"qgallouedec/lge","url":"https://github.com/qgallouedec/lge"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-alien","task":"Atari Games","dataset_variant":"Atari 2600 Alien","rows":49,"metrics":["Score"],"first_row_in_archive_order":{"model":"MuZero","paper":"/paper/mastering-atari-go-chess-and-shogi-by","metrics":{"Score":"741812.63"},"code_links":[{"title":"werner-duvaud/muzero-general","url":"https://github.com/werner-duvaud/muzero-general"},{"title":"opendilab/LightZero","url":"https://github.com/opendilab/LightZero"},{"title":"koulanurag/muzero-pytorch","url":"https://github.com/koulanurag/muzero-pytorch"},{"title":"johan-gras/MuZero","url":"https://github.com/johan-gras/MuZero"},{"title":"kaesve/muzero","url":"https://github.com/kaesve/muzero"},{"title":"Zeta36/muzero","url":"https://github.com/Zeta36/muzero"},{"title":"YuriCat/MuZeroJupyterExample","url":"https://github.com/YuriCat/MuZeroJupyterExample"},{"title":"DHDev0/Muzero","url":"https://github.com/DHDev0/Muzero"},{"title":"foersterrobert/MuZero","url":"https://github.com/foersterrobert/MuZero"},{"title":"k-lombard/CS4641_Project","url":"https://github.com/k-lombard/CS4641_Project"},{"title":"k-lombard/Deep-Learning-Chess-AI","url":"https://github.com/k-lombard/Deep-Learning-Chess-AI"},{"title":"JuanCCS/muzero-jc","url":"https://github.com/JuanCCS/muzero-jc"},{"title":"ZiyuanMa/reversi","url":"https://github.com/ZiyuanMa/reversi"},{"title":"Miatto-research-group/muzero","url":"https://github.com/Miatto-research-group/muzero"},{"title":"colindbrown/columbia-deep-learning-project","url":"https://github.com/colindbrown/columbia-deep-learning-project"},{"title":"SHRIVP/muzero","url":"https://github.com/SHRIVP/muzero"},{"title":"dmiracle/muzero-starter","url":"https://github.com/dmiracle/muzero-starter"},{"title":"snjstudent/MyMuzero","url":"https://github.com/snjstudent/MyMuzero"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-beam-rider","task":"Atari Games","dataset_variant":"Atari 2600 Beam Rider","rows":49,"metrics":["Score","Return"],"first_row_in_archive_order":{"model":"MuZero","paper":"/paper/mastering-atari-go-chess-and-shogi-by","metrics":{"Score":"454993.53"},"code_links":[{"title":"werner-duvaud/muzero-general","url":"https://github.com/werner-duvaud/muzero-general"},{"title":"opendilab/LightZero","url":"https://github.com/opendilab/LightZero"},{"title":"koulanurag/muzero-pytorch","url":"https://github.com/koulanurag/muzero-pytorch"},{"title":"johan-gras/MuZero","url":"https://github.com/johan-gras/MuZero"},{"title":"kaesve/muzero","url":"https://github.com/kaesve/muzero"},{"title":"Zeta36/muzero","url":"https://github.com/Zeta36/muzero"},{"title":"YuriCat/MuZeroJupyterExample","url":"https://github.com/YuriCat/MuZeroJupyterExample"},{"title":"DHDev0/Muzero","url":"https://github.com/DHDev0/Muzero"},{"title":"foersterrobert/MuZero","url":"https://github.com/foersterrobert/MuZero"},{"title":"k-lombard/CS4641_Project","url":"https://github.com/k-lombard/CS4641_Project"},{"title":"k-lombard/Deep-Learning-Chess-AI","url":"https://github.com/k-lombard/Deep-Learning-Chess-AI"},{"title":"JuanCCS/muzero-jc","url":"https://github.com/JuanCCS/muzero-jc"},{"title":"ZiyuanMa/reversi","url":"https://github.com/ZiyuanMa/reversi"},{"title":"Miatto-research-group/muzero","url":"https://github.com/Miatto-research-group/muzero"},{"title":"colindbrown/columbia-deep-learning-project","url":"https://github.com/colindbrown/columbia-deep-learning-project"},{"title":"SHRIVP/muzero","url":"https://github.com/SHRIVP/muzero"},{"title":"dmiracle/muzero-starter","url":"https://github.com/dmiracle/muzero-starter"},{"title":"snjstudent/MyMuzero","url":"https://github.com/snjstudent/MyMuzero"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-crazy-climber","task":"Atari Games","dataset_variant":"Atari 2600 Crazy Climber","rows":49,"metrics":["Score"],"first_row_in_archive_order":{"model":"Agent57","paper":"/paper/agent57-outperforming-the-atari-human","metrics":{"Score":"565909.85"},"code_links":[{"title":"michaelnny/deep_rl_zoo","url":"https://github.com/michaelnny/deep_rl_zoo"},{"title":"pocokhc/agent57","url":"https://github.com/pocokhc/agent57"},{"title":"yuta0821/agent57_pytorch","url":"https://github.com/yuta0821/agent57_pytorch"},{"title":"Nkluge-correa/teeny-tiny_castle","url":"https://github.com/Nkluge-correa/teeny-tiny_castle"},{"title":"YHL04/agent57","url":"https://github.com/YHL04/agent57"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-amidar","task":"Atari Games","dataset_variant":"Atari 2600 Amidar","rows":48,"metrics":["Score"],"first_row_in_archive_order":{"model":"Agent57","paper":"/paper/agent57-outperforming-the-atari-human","metrics":{"Score":"29660.08"},"code_links":[{"title":"michaelnny/deep_rl_zoo","url":"https://github.com/michaelnny/deep_rl_zoo"},{"title":"pocokhc/agent57","url":"https://github.com/pocokhc/agent57"},{"title":"yuta0821/agent57_pytorch","url":"https://github.com/yuta0821/agent57_pytorch"},{"title":"Nkluge-correa/teeny-tiny_castle","url":"https://github.com/Nkluge-correa/teeny-tiny_castle"},{"title":"YHL04/agent57","url":"https://github.com/YHL04/agent57"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-battle-zone","task":"Atari Games","dataset_variant":"Atari 2600 Battle Zone","rows":47,"metrics":["Score"],"first_row_in_archive_order":{"model":"Agent57","paper":"/paper/agent57-outperforming-the-atari-human","metrics":{"Score":"934134.88"},"code_links":[{"title":"michaelnny/deep_rl_zoo","url":"https://github.com/michaelnny/deep_rl_zoo"},{"title":"pocokhc/agent57","url":"https://github.com/pocokhc/agent57"},{"title":"yuta0821/agent57_pytorch","url":"https://github.com/yuta0821/agent57_pytorch"},{"title":"Nkluge-correa/teeny-tiny_castle","url":"https://github.com/Nkluge-correa/teeny-tiny_castle"},{"title":"YHL04/agent57","url":"https://github.com/YHL04/agent57"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-kangaroo","task":"Atari Games","dataset_variant":"Atari 2600 Kangaroo","rows":47,"metrics":["Score"],"first_row_in_archive_order":{"model":"Agent57","paper":"/paper/agent57-outperforming-the-atari-human","metrics":{"Score":"24034.16"},"code_links":[{"title":"michaelnny/deep_rl_zoo","url":"https://github.com/michaelnny/deep_rl_zoo"},{"title":"pocokhc/agent57","url":"https://github.com/pocokhc/agent57"},{"title":"yuta0821/agent57_pytorch","url":"https://github.com/yuta0821/agent57_pytorch"},{"title":"Nkluge-correa/teeny-tiny_castle","url":"https://github.com/Nkluge-correa/teeny-tiny_castle"},{"title":"YHL04/agent57","url":"https://github.com/YHL04/agent57"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-ms-pacman","task":"Atari Games","dataset_variant":"Atari 2600 Ms. Pacman","rows":47,"metrics":["Score","Best Score"],"first_row_in_archive_order":{"model":"MuZero","paper":"/paper/mastering-atari-go-chess-and-shogi-by","metrics":{"Score":"243401.10"},"code_links":[{"title":"werner-duvaud/muzero-general","url":"https://github.com/werner-duvaud/muzero-general"},{"title":"opendilab/LightZero","url":"https://github.com/opendilab/LightZero"},{"title":"koulanurag/muzero-pytorch","url":"https://github.com/koulanurag/muzero-pytorch"},{"title":"johan-gras/MuZero","url":"https://github.com/johan-gras/MuZero"},{"title":"kaesve/muzero","url":"https://github.com/kaesve/muzero"},{"title":"Zeta36/muzero","url":"https://github.com/Zeta36/muzero"},{"title":"YuriCat/MuZeroJupyterExample","url":"https://github.com/YuriCat/MuZeroJupyterExample"},{"title":"DHDev0/Muzero","url":"https://github.com/DHDev0/Muzero"},{"title":"foersterrobert/MuZero","url":"https://github.com/foersterrobert/MuZero"},{"title":"k-lombard/CS4641_Project","url":"https://github.com/k-lombard/CS4641_Project"},{"title":"k-lombard/Deep-Learning-Chess-AI","url":"https://github.com/k-lombard/Deep-Learning-Chess-AI"},{"title":"JuanCCS/muzero-jc","url":"https://github.com/JuanCCS/muzero-jc"},{"title":"ZiyuanMa/reversi","url":"https://github.com/ZiyuanMa/reversi"},{"title":"Miatto-research-group/muzero","url":"https://github.com/Miatto-research-group/muzero"},{"title":"colindbrown/columbia-deep-learning-project","url":"https://github.com/colindbrown/columbia-deep-learning-project"},{"title":"SHRIVP/muzero","url":"https://github.com/SHRIVP/muzero"},{"title":"dmiracle/muzero-starter","url":"https://github.com/dmiracle/muzero-starter"},{"title":"snjstudent/MyMuzero","url":"https://github.com/snjstudent/MyMuzero"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-demon-attack","task":"Atari Games","dataset_variant":"Atari 2600 Demon Attack","rows":46,"metrics":["Score"],"first_row_in_archive_order":{"model":"GDI-H3","paper":"/paper/generalized-data-distribution-iteration","metrics":{"Score":"787985"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-assault","task":"Atari Games","dataset_variant":"Atari 2600 Assault","rows":45,"metrics":["Score"],"first_row_in_archive_order":{"model":"MuZero","paper":"/paper/mastering-atari-go-chess-and-shogi-by","metrics":{"Score":"143972.03"},"code_links":[{"title":"werner-duvaud/muzero-general","url":"https://github.com/werner-duvaud/muzero-general"},{"title":"opendilab/LightZero","url":"https://github.com/opendilab/LightZero"},{"title":"koulanurag/muzero-pytorch","url":"https://github.com/koulanurag/muzero-pytorch"},{"title":"johan-gras/MuZero","url":"https://github.com/johan-gras/MuZero"},{"title":"kaesve/muzero","url":"https://github.com/kaesve/muzero"},{"title":"Zeta36/muzero","url":"https://github.com/Zeta36/muzero"},{"title":"YuriCat/MuZeroJupyterExample","url":"https://github.com/YuriCat/MuZeroJupyterExample"},{"title":"DHDev0/Muzero","url":"https://github.com/DHDev0/Muzero"},{"title":"foersterrobert/MuZero","url":"https://github.com/foersterrobert/MuZero"},{"title":"k-lombard/CS4641_Project","url":"https://github.com/k-lombard/CS4641_Project"},{"title":"k-lombard/Deep-Learning-Chess-AI","url":"https://github.com/k-lombard/Deep-Learning-Chess-AI"},{"title":"JuanCCS/muzero-jc","url":"https://github.com/JuanCCS/muzero-jc"},{"title":"ZiyuanMa/reversi","url":"https://github.com/ZiyuanMa/reversi"},{"title":"Miatto-research-group/muzero","url":"https://github.com/Miatto-research-group/muzero"},{"title":"colindbrown/columbia-deep-learning-project","url":"https://github.com/colindbrown/columbia-deep-learning-project"},{"title":"SHRIVP/muzero","url":"https://github.com/SHRIVP/muzero"},{"title":"dmiracle/muzero-starter","url":"https://github.com/dmiracle/muzero-starter"},{"title":"snjstudent/MyMuzero","url":"https://github.com/snjstudent/MyMuzero"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-bank-heist","task":"Atari Games","dataset_variant":"Atari 2600 Bank Heist","rows":45,"metrics":["Score"],"first_row_in_archive_order":{"model":"MuZero (Res2 Adam)","paper":"/paper/online-and-offline-reinforcement-learning-by","metrics":{"Score":"27219.8"},"code_links":[{"title":"DHDev0/Muzero-unplugged","url":"https://github.com/DHDev0/Muzero-unplugged"},{"title":"enpasos/muzero","url":"https://github.com/enpasos/muzero"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-centipede","task":"Atari Games","dataset_variant":"Atari 2600 Centipede","rows":45,"metrics":["Score"],"first_row_in_archive_order":{"model":"Go-Explore","paper":"/paper/first-return-then-explore","metrics":{"Score":"1422628"},"code_links":[{"title":"uber-research/go-explore","url":"https://github.com/uber-research/go-explore"},{"title":"qgallouedec/lge","url":"https://github.com/qgallouedec/lge"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-chopper-command","task":"Atari Games","dataset_variant":"Atari 2600 Chopper Command","rows":45,"metrics":["Score"],"first_row_in_archive_order":{"model":"GDI-H3","paper":"/paper/gdi-rethinking-what-makes-reinforcement","metrics":{"Score":"999999"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-james-bond","task":"Atari Games","dataset_variant":"Atari 2600 James Bond","rows":45,"metrics":["Score","Medium Human-Normalized Score"],"first_row_in_archive_order":{"model":"GDI-H3","paper":"/paper/generalized-data-distribution-iteration","metrics":{"Score":"620780"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-krull","task":"Atari Games","dataset_variant":"Atari 2600 Krull","rows":45,"metrics":["Score"],"first_row_in_archive_order":{"model":"GDI-H3","paper":"/paper/generalized-data-distribution-iteration","metrics":{"Score":"594540"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-bowling","task":"Atari Games","dataset_variant":"Atari 2600 Bowling","rows":44,"metrics":["Score"],"first_row_in_archive_order":{"model":"MuZero","paper":"/paper/mastering-atari-go-chess-and-shogi-by","metrics":{"Score":"260.13"},"code_links":[{"title":"werner-duvaud/muzero-general","url":"https://github.com/werner-duvaud/muzero-general"},{"title":"opendilab/LightZero","url":"https://github.com/opendilab/LightZero"},{"title":"koulanurag/muzero-pytorch","url":"https://github.com/koulanurag/muzero-pytorch"},{"title":"johan-gras/MuZero","url":"https://github.com/johan-gras/MuZero"},{"title":"kaesve/muzero","url":"https://github.com/kaesve/muzero"},{"title":"Zeta36/muzero","url":"https://github.com/Zeta36/muzero"},{"title":"YuriCat/MuZeroJupyterExample","url":"https://github.com/YuriCat/MuZeroJupyterExample"},{"title":"DHDev0/Muzero","url":"https://github.com/DHDev0/Muzero"},{"title":"foersterrobert/MuZero","url":"https://github.com/foersterrobert/MuZero"},{"title":"k-lombard/CS4641_Project","url":"https://github.com/k-lombard/CS4641_Project"},{"title":"k-lombard/Deep-Learning-Chess-AI","url":"https://github.com/k-lombard/Deep-Learning-Chess-AI"},{"title":"JuanCCS/muzero-jc","url":"https://github.com/JuanCCS/muzero-jc"},{"title":"ZiyuanMa/reversi","url":"https://github.com/ZiyuanMa/reversi"},{"title":"Miatto-research-group/muzero","url":"https://github.com/Miatto-research-group/muzero"},{"title":"colindbrown/columbia-deep-learning-project","url":"https://github.com/colindbrown/columbia-deep-learning-project"},{"title":"SHRIVP/muzero","url":"https://github.com/SHRIVP/muzero"},{"title":"dmiracle/muzero-starter","url":"https://github.com/dmiracle/muzero-starter"},{"title":"snjstudent/MyMuzero","url":"https://github.com/snjstudent/MyMuzero"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-fishing-derby","task":"Atari Games","dataset_variant":"Atari 2600 Fishing Derby","rows":44,"metrics":["Score"],"first_row_in_archive_order":{"model":"MuZero","paper":"/paper/mastering-atari-go-chess-and-shogi-by","metrics":{"Score":"91.16"},"code_links":[{"title":"werner-duvaud/muzero-general","url":"https://github.com/werner-duvaud/muzero-general"},{"title":"opendilab/LightZero","url":"https://github.com/opendilab/LightZero"},{"title":"koulanurag/muzero-pytorch","url":"https://github.com/koulanurag/muzero-pytorch"},{"title":"johan-gras/MuZero","url":"https://github.com/johan-gras/MuZero"},{"title":"kaesve/muzero","url":"https://github.com/kaesve/muzero"},{"title":"Zeta36/muzero","url":"https://github.com/Zeta36/muzero"},{"title":"YuriCat/MuZeroJupyterExample","url":"https://github.com/YuriCat/MuZeroJupyterExample"},{"title":"DHDev0/Muzero","url":"https://github.com/DHDev0/Muzero"},{"title":"foersterrobert/MuZero","url":"https://github.com/foersterrobert/MuZero"},{"title":"k-lombard/CS4641_Project","url":"https://github.com/k-lombard/CS4641_Project"},{"title":"k-lombard/Deep-Learning-Chess-AI","url":"https://github.com/k-lombard/Deep-Learning-Chess-AI"},{"title":"JuanCCS/muzero-jc","url":"https://github.com/JuanCCS/muzero-jc"},{"title":"ZiyuanMa/reversi","url":"https://github.com/ZiyuanMa/reversi"},{"title":"Miatto-research-group/muzero","url":"https://github.com/Miatto-research-group/muzero"},{"title":"colindbrown/columbia-deep-learning-project","url":"https://github.com/colindbrown/columbia-deep-learning-project"},{"title":"SHRIVP/muzero","url":"https://github.com/SHRIVP/muzero"},{"title":"dmiracle/muzero-starter","url":"https://github.com/dmiracle/muzero-starter"},{"title":"snjstudent/MyMuzero","url":"https://github.com/snjstudent/MyMuzero"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-hero","task":"Atari Games","dataset_variant":"Atari 2600 HERO","rows":44,"metrics":["Score","Best Score"],"first_row_in_archive_order":{"model":"Agent57","paper":"/paper/agent57-outperforming-the-atari-human","metrics":{"Score":"114736.26"},"code_links":[{"title":"michaelnny/deep_rl_zoo","url":"https://github.com/michaelnny/deep_rl_zoo"},{"title":"pocokhc/agent57","url":"https://github.com/pocokhc/agent57"},{"title":"yuta0821/agent57_pytorch","url":"https://github.com/yuta0821/agent57_pytorch"},{"title":"Nkluge-correa/teeny-tiny_castle","url":"https://github.com/Nkluge-correa/teeny-tiny_castle"},{"title":"YHL04/agent57","url":"https://github.com/YHL04/agent57"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-road-runner","task":"Atari Games","dataset_variant":"Atari 2600 Road Runner","rows":44,"metrics":["Score"],"first_row_in_archive_order":{"model":"GDI-H3","paper":"/paper/generalized-data-distribution-iteration","metrics":{"Score":"999999"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-time-pilot","task":"Atari Games","dataset_variant":"Atari 2600 Time Pilot","rows":44,"metrics":["Score"],"first_row_in_archive_order":{"model":"MuZero","paper":"/paper/mastering-atari-go-chess-and-shogi-by","metrics":{"Score":"476763.90"},"code_links":[{"title":"werner-duvaud/muzero-general","url":"https://github.com/werner-duvaud/muzero-general"},{"title":"opendilab/LightZero","url":"https://github.com/opendilab/LightZero"},{"title":"koulanurag/muzero-pytorch","url":"https://github.com/koulanurag/muzero-pytorch"},{"title":"johan-gras/MuZero","url":"https://github.com/johan-gras/MuZero"},{"title":"kaesve/muzero","url":"https://github.com/kaesve/muzero"},{"title":"Zeta36/muzero","url":"https://github.com/Zeta36/muzero"},{"title":"YuriCat/MuZeroJupyterExample","url":"https://github.com/YuriCat/MuZeroJupyterExample"},{"title":"DHDev0/Muzero","url":"https://github.com/DHDev0/Muzero"},{"title":"foersterrobert/MuZero","url":"https://github.com/foersterrobert/MuZero"},{"title":"k-lombard/CS4641_Project","url":"https://github.com/k-lombard/CS4641_Project"},{"title":"k-lombard/Deep-Learning-Chess-AI","url":"https://github.com/k-lombard/Deep-Learning-Chess-AI"},{"title":"JuanCCS/muzero-jc","url":"https://github.com/JuanCCS/muzero-jc"},{"title":"ZiyuanMa/reversi","url":"https://github.com/ZiyuanMa/reversi"},{"title":"Miatto-research-group/muzero","url":"https://github.com/Miatto-research-group/muzero"},{"title":"colindbrown/columbia-deep-learning-project","url":"https://github.com/colindbrown/columbia-deep-learning-project"},{"title":"SHRIVP/muzero","url":"https://github.com/SHRIVP/muzero"},{"title":"dmiracle/muzero-starter","url":"https://github.com/dmiracle/muzero-starter"},{"title":"snjstudent/MyMuzero","url":"https://github.com/snjstudent/MyMuzero"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-tutankham","task":"Atari Games","dataset_variant":"Atari 2600 Tutankham","rows":44,"metrics":["Score"],"first_row_in_archive_order":{"model":"Agent57","paper":"/paper/agent57-outperforming-the-atari-human","metrics":{"Score":"2354.91"},"code_links":[{"title":"michaelnny/deep_rl_zoo","url":"https://github.com/michaelnny/deep_rl_zoo"},{"title":"pocokhc/agent57","url":"https://github.com/pocokhc/agent57"},{"title":"yuta0821/agent57_pytorch","url":"https://github.com/yuta0821/agent57_pytorch"},{"title":"Nkluge-correa/teeny-tiny_castle","url":"https://github.com/Nkluge-correa/teeny-tiny_castle"},{"title":"YHL04/agent57","url":"https://github.com/YHL04/agent57"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-up-and-down","task":"Atari Games","dataset_variant":"Atari 2600 Up and Down","rows":44,"metrics":["Score"],"first_row_in_archive_order":{"model":"GDI-I3","paper":"/paper/gdi-rethinking-what-makes-reinforcement","metrics":{"Score":"986440"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-asteroids","task":"Atari Games","dataset_variant":"Atari 2600 Asteroids","rows":43,"metrics":["Score"],"first_row_in_archive_order":{"model":"GDI-H3","paper":"/paper/generalized-data-distribution-iteration","metrics":{"Score":"760005"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-double-dunk","task":"Atari Games","dataset_variant":"Atari 2600 Double Dunk","rows":43,"metrics":["Score"],"first_row_in_archive_order":{"model":"UCT","paper":"/paper/the-arcade-learning-environment-an-evaluation","metrics":{"Score":"24"},"code_links":[{"title":"mgbellemare/Arcade-Learning-Environment","url":"https://github.com/mgbellemare/Arcade-Learning-Environment"},{"title":"kenjyoung/MinAtar","url":"https://github.com/kenjyoung/MinAtar"},{"title":"nandomp/AICollaboratory","url":"https://github.com/nandomp/AICollaboratory"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-gopher","task":"Atari Games","dataset_variant":"Atari 2600 Gopher","rows":43,"metrics":["Score"],"first_row_in_archive_order":{"model":"GDI-I3","paper":"/paper/generalized-data-distribution-iteration","metrics":{"Score":"488830"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-ice-hockey","task":"Atari Games","dataset_variant":"Atari 2600 Ice Hockey","rows":43,"metrics":["Score"],"first_row_in_archive_order":{"model":"GDI-H3","paper":"/paper/generalized-data-distribution-iteration","metrics":{"Score":"481.9"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-kung-fu-master","task":"Atari Games","dataset_variant":"Atari 2600 Kung-Fu Master","rows":43,"metrics":["Score"],"first_row_in_archive_order":{"model":"GDI-H3","paper":"/paper/generalized-data-distribution-iteration","metrics":{"Score":"1666665"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-name-this-game","task":"Atari Games","dataset_variant":"Atari 2600 Name This Game","rows":43,"metrics":["Score"],"first_row_in_archive_order":{"model":"MuZero","paper":"/paper/mastering-atari-go-chess-and-shogi-by","metrics":{"Score":"157177.85"},"code_links":[{"title":"werner-duvaud/muzero-general","url":"https://github.com/werner-duvaud/muzero-general"},{"title":"opendilab/LightZero","url":"https://github.com/opendilab/LightZero"},{"title":"koulanurag/muzero-pytorch","url":"https://github.com/koulanurag/muzero-pytorch"},{"title":"johan-gras/MuZero","url":"https://github.com/johan-gras/MuZero"},{"title":"kaesve/muzero","url":"https://github.com/kaesve/muzero"},{"title":"Zeta36/muzero","url":"https://github.com/Zeta36/muzero"},{"title":"YuriCat/MuZeroJupyterExample","url":"https://github.com/YuriCat/MuZeroJupyterExample"},{"title":"DHDev0/Muzero","url":"https://github.com/DHDev0/Muzero"},{"title":"foersterrobert/MuZero","url":"https://github.com/foersterrobert/MuZero"},{"title":"k-lombard/CS4641_Project","url":"https://github.com/k-lombard/CS4641_Project"},{"title":"k-lombard/Deep-Learning-Chess-AI","url":"https://github.com/k-lombard/Deep-Learning-Chess-AI"},{"title":"JuanCCS/muzero-jc","url":"https://github.com/JuanCCS/muzero-jc"},{"title":"ZiyuanMa/reversi","url":"https://github.com/ZiyuanMa/reversi"},{"title":"Miatto-research-group/muzero","url":"https://github.com/Miatto-research-group/muzero"},{"title":"colindbrown/columbia-deep-learning-project","url":"https://github.com/colindbrown/columbia-deep-learning-project"},{"title":"SHRIVP/muzero","url":"https://github.com/SHRIVP/muzero"},{"title":"dmiracle/muzero-starter","url":"https://github.com/dmiracle/muzero-starter"},{"title":"snjstudent/MyMuzero","url":"https://github.com/snjstudent/MyMuzero"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-tennis","task":"Atari Games","dataset_variant":"Atari 2600 Tennis","rows":43,"metrics":["Score"],"first_row_in_archive_order":{"model":"GDI-I3","paper":"/paper/gdi-rethinking-what-makes-reinforcement","metrics":{"Score":"24"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-atlantis","task":"Atari Games","dataset_variant":"Atari 2600 Atlantis","rows":42,"metrics":["Score"],"first_row_in_archive_order":{"model":"GDI-H3","paper":"/paper/generalized-data-distribution-iteration","metrics":{"Score":"3837300"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-robotank","task":"Atari Games","dataset_variant":"Atari 2600 Robotank","rows":42,"metrics":["Score"],"first_row_in_archive_order":{"model":"MuZero","paper":"/paper/mastering-atari-go-chess-and-shogi-by","metrics":{"Score":"131.13"},"code_links":[{"title":"werner-duvaud/muzero-general","url":"https://github.com/werner-duvaud/muzero-general"},{"title":"opendilab/LightZero","url":"https://github.com/opendilab/LightZero"},{"title":"koulanurag/muzero-pytorch","url":"https://github.com/koulanurag/muzero-pytorch"},{"title":"johan-gras/MuZero","url":"https://github.com/johan-gras/MuZero"},{"title":"kaesve/muzero","url":"https://github.com/kaesve/muzero"},{"title":"Zeta36/muzero","url":"https://github.com/Zeta36/muzero"},{"title":"YuriCat/MuZeroJupyterExample","url":"https://github.com/YuriCat/MuZeroJupyterExample"},{"title":"DHDev0/Muzero","url":"https://github.com/DHDev0/Muzero"},{"title":"foersterrobert/MuZero","url":"https://github.com/foersterrobert/MuZero"},{"title":"k-lombard/CS4641_Project","url":"https://github.com/k-lombard/CS4641_Project"},{"title":"k-lombard/Deep-Learning-Chess-AI","url":"https://github.com/k-lombard/Deep-Learning-Chess-AI"},{"title":"JuanCCS/muzero-jc","url":"https://github.com/JuanCCS/muzero-jc"},{"title":"ZiyuanMa/reversi","url":"https://github.com/ZiyuanMa/reversi"},{"title":"Miatto-research-group/muzero","url":"https://github.com/Miatto-research-group/muzero"},{"title":"colindbrown/columbia-deep-learning-project","url":"https://github.com/colindbrown/columbia-deep-learning-project"},{"title":"SHRIVP/muzero","url":"https://github.com/SHRIVP/muzero"},{"title":"dmiracle/muzero-starter","url":"https://github.com/dmiracle/muzero-starter"},{"title":"snjstudent/MyMuzero","url":"https://github.com/snjstudent/MyMuzero"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-video-pinball","task":"Atari Games","dataset_variant":"Atari 2600 Video Pinball","rows":42,"metrics":["Score"],"first_row_in_archive_order":{"model":"R2D2","paper":"/paper/recurrent-experience-replay-in-distributed","metrics":{"Score":"999383.2"},"code_links":[{"title":"opendilab/DI-engine","url":"https://github.com/opendilab/DI-engine/blob/main/ding/policy/r2d2.py"},{"title":"michaelnny/deep_rl_zoo","url":"https://github.com/michaelnny/deep_rl_zoo"},{"title":"garymm/earl","url":"https://github.com/garymm/earl/tree/master/earl/agents/r2d2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-river-raid","task":"Atari Games","dataset_variant":"Atari 2600 River Raid","rows":41,"metrics":["Score","Best Score"],"first_row_in_archive_order":{"model":"MuZero","paper":"/paper/mastering-atari-go-chess-and-shogi-by","metrics":{"Score":"323417.18"},"code_links":[{"title":"werner-duvaud/muzero-general","url":"https://github.com/werner-duvaud/muzero-general"},{"title":"opendilab/LightZero","url":"https://github.com/opendilab/LightZero"},{"title":"koulanurag/muzero-pytorch","url":"https://github.com/koulanurag/muzero-pytorch"},{"title":"johan-gras/MuZero","url":"https://github.com/johan-gras/MuZero"},{"title":"kaesve/muzero","url":"https://github.com/kaesve/muzero"},{"title":"Zeta36/muzero","url":"https://github.com/Zeta36/muzero"},{"title":"YuriCat/MuZeroJupyterExample","url":"https://github.com/YuriCat/MuZeroJupyterExample"},{"title":"DHDev0/Muzero","url":"https://github.com/DHDev0/Muzero"},{"title":"foersterrobert/MuZero","url":"https://github.com/foersterrobert/MuZero"},{"title":"k-lombard/CS4641_Project","url":"https://github.com/k-lombard/CS4641_Project"},{"title":"k-lombard/Deep-Learning-Chess-AI","url":"https://github.com/k-lombard/Deep-Learning-Chess-AI"},{"title":"JuanCCS/muzero-jc","url":"https://github.com/JuanCCS/muzero-jc"},{"title":"ZiyuanMa/reversi","url":"https://github.com/ZiyuanMa/reversi"},{"title":"Miatto-research-group/muzero","url":"https://github.com/Miatto-research-group/muzero"},{"title":"colindbrown/columbia-deep-learning-project","url":"https://github.com/colindbrown/columbia-deep-learning-project"},{"title":"SHRIVP/muzero","url":"https://github.com/SHRIVP/muzero"},{"title":"dmiracle/muzero-starter","url":"https://github.com/dmiracle/muzero-starter"},{"title":"snjstudent/MyMuzero","url":"https://github.com/snjstudent/MyMuzero"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-wizard-of-wor","task":"Atari Games","dataset_variant":"Atari 2600 Wizard of Wor","rows":41,"metrics":["Score"],"first_row_in_archive_order":{"model":"MuZero","paper":"/paper/mastering-atari-go-chess-and-shogi-by","metrics":{"Score":"197126.00"},"code_links":[{"title":"werner-duvaud/muzero-general","url":"https://github.com/werner-duvaud/muzero-general"},{"title":"opendilab/LightZero","url":"https://github.com/opendilab/LightZero"},{"title":"koulanurag/muzero-pytorch","url":"https://github.com/koulanurag/muzero-pytorch"},{"title":"johan-gras/MuZero","url":"https://github.com/johan-gras/MuZero"},{"title":"kaesve/muzero","url":"https://github.com/kaesve/muzero"},{"title":"Zeta36/muzero","url":"https://github.com/Zeta36/muzero"},{"title":"YuriCat/MuZeroJupyterExample","url":"https://github.com/YuriCat/MuZeroJupyterExample"},{"title":"DHDev0/Muzero","url":"https://github.com/DHDev0/Muzero"},{"title":"foersterrobert/MuZero","url":"https://github.com/foersterrobert/MuZero"},{"title":"k-lombard/CS4641_Project","url":"https://github.com/k-lombard/CS4641_Project"},{"title":"k-lombard/Deep-Learning-Chess-AI","url":"https://github.com/k-lombard/Deep-Learning-Chess-AI"},{"title":"JuanCCS/muzero-jc","url":"https://github.com/JuanCCS/muzero-jc"},{"title":"ZiyuanMa/reversi","url":"https://github.com/ZiyuanMa/reversi"},{"title":"Miatto-research-group/muzero","url":"https://github.com/Miatto-research-group/muzero"},{"title":"colindbrown/columbia-deep-learning-project","url":"https://github.com/colindbrown/columbia-deep-learning-project"},{"title":"SHRIVP/muzero","url":"https://github.com/SHRIVP/muzero"},{"title":"dmiracle/muzero-starter","url":"https://github.com/dmiracle/muzero-starter"},{"title":"snjstudent/MyMuzero","url":"https://github.com/snjstudent/MyMuzero"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-zaxxon","task":"Atari Games","dataset_variant":"Atari 2600 Zaxxon","rows":41,"metrics":["Score"],"first_row_in_archive_order":{"model":"MuZero","paper":"/paper/mastering-atari-go-chess-and-shogi-by","metrics":{"Score":"725853.90"},"code_links":[{"title":"werner-duvaud/muzero-general","url":"https://github.com/werner-duvaud/muzero-general"},{"title":"opendilab/LightZero","url":"https://github.com/opendilab/LightZero"},{"title":"koulanurag/muzero-pytorch","url":"https://github.com/koulanurag/muzero-pytorch"},{"title":"johan-gras/MuZero","url":"https://github.com/johan-gras/MuZero"},{"title":"kaesve/muzero","url":"https://github.com/kaesve/muzero"},{"title":"Zeta36/muzero","url":"https://github.com/Zeta36/muzero"},{"title":"YuriCat/MuZeroJupyterExample","url":"https://github.com/YuriCat/MuZeroJupyterExample"},{"title":"DHDev0/Muzero","url":"https://github.com/DHDev0/Muzero"},{"title":"foersterrobert/MuZero","url":"https://github.com/foersterrobert/MuZero"},{"title":"k-lombard/CS4641_Project","url":"https://github.com/k-lombard/CS4641_Project"},{"title":"k-lombard/Deep-Learning-Chess-AI","url":"https://github.com/k-lombard/Deep-Learning-Chess-AI"},{"title":"JuanCCS/muzero-jc","url":"https://github.com/JuanCCS/muzero-jc"},{"title":"ZiyuanMa/reversi","url":"https://github.com/ZiyuanMa/reversi"},{"title":"Miatto-research-group/muzero","url":"https://github.com/Miatto-research-group/muzero"},{"title":"colindbrown/columbia-deep-learning-project","url":"https://github.com/colindbrown/columbia-deep-learning-project"},{"title":"SHRIVP/muzero","url":"https://github.com/SHRIVP/muzero"},{"title":"dmiracle/muzero-starter","url":"https://github.com/dmiracle/muzero-starter"},{"title":"snjstudent/MyMuzero","url":"https://github.com/snjstudent/MyMuzero"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-berzerk","task":"Atari Games","dataset_variant":"Atari 2600 Berzerk","rows":39,"metrics":["Score"],"first_row_in_archive_order":{"model":"Go-Explore","paper":"/paper/first-return-then-explore","metrics":{"Score":"197376"},"code_links":[{"title":"uber-research/go-explore","url":"https://github.com/uber-research/go-explore"},{"title":"qgallouedec/lge","url":"https://github.com/qgallouedec/lge"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-pitfall","task":"Atari Games","dataset_variant":"Atari 2600 Pitfall!","rows":23,"metrics":["Score"],"first_row_in_archive_order":{"model":"Go-Explore","paper":"/paper/go-explore-a-new-approach-for-hard","metrics":{"Score":"102571"},"code_links":[{"title":"uber-research/go-explore","url":"https://github.com/uber-research/go-explore"},{"title":"Adeikalam/Go-Explore","url":"https://github.com/Adeikalam/Go-Explore"},{"title":"Dorozhko-Anton/go-explore","url":"https://github.com/Dorozhko-Anton/go-explore"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-skiing","task":"Atari Games","dataset_variant":"Atari 2600 Skiing","rows":23,"metrics":["Score"],"first_row_in_archive_order":{"model":"Best Learner","paper":"/paper/the-arcade-learning-environment-an-evaluation","metrics":{"Score":"0"},"code_links":[{"title":"mgbellemare/Arcade-Learning-Environment","url":"https://github.com/mgbellemare/Arcade-Learning-Environment"},{"title":"kenjyoung/MinAtar","url":"https://github.com/kenjyoung/MinAtar"},{"title":"nandomp/AICollaboratory","url":"https://github.com/nandomp/AICollaboratory"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-phoenix","task":"Atari Games","dataset_variant":"Atari 2600 Phoenix","rows":21,"metrics":["Score"],"first_row_in_archive_order":{"model":"GDI-H3","paper":"/paper/generalized-data-distribution-iteration","metrics":{"Score":"959580"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-yars-revenge","task":"Atari Games","dataset_variant":"Atari 2600 Yars Revenge","rows":17,"metrics":["Score"],"first_row_in_archive_order":{"model":"Agent57","paper":"/paper/agent57-outperforming-the-atari-human","metrics":{"Score":"998532.37"},"code_links":[{"title":"michaelnny/deep_rl_zoo","url":"https://github.com/michaelnny/deep_rl_zoo"},{"title":"pocokhc/agent57","url":"https://github.com/pocokhc/agent57"},{"title":"yuta0821/agent57_pytorch","url":"https://github.com/yuta0821/agent57_pytorch"},{"title":"Nkluge-correa/teeny-tiny_castle","url":"https://github.com/Nkluge-correa/teeny-tiny_castle"},{"title":"YHL04/agent57","url":"https://github.com/YHL04/agent57"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-surround","task":"Atari Games","dataset_variant":"Atari 2600 Surround","rows":15,"metrics":["Score"],"first_row_in_archive_order":{"model":"NoisyNet-Dueling","paper":"/paper/noisy-networks-for-exploration","metrics":{"Score":"10"},"code_links":[{"title":"opendilab/DI-engine","url":"https://github.com/opendilab/DI-engine"},{"title":"Curt-Park/rainbow-is-all-you-need","url":"https://github.com/Curt-Park/rainbow-is-all-you-need"},{"title":"chainer/chainerrl","url":"https://github.com/chainer/chainerrl"},{"title":"Kaixhin/NoisyNet-A3C","url":"https://github.com/Kaixhin/NoisyNet-A3C"},{"title":"BY571/DQN-Atari-Agents","url":"https://github.com/BY571/DQN-Atari-Agents"},{"title":"chandar-lab/RLHive","url":"https://github.com/chandar-lab/RLHive"},{"title":"seungjaeryanlee/rldb","url":"https://github.com/seungjaeryanlee/rldb"},{"title":"behzadanksu/rl-attack","url":"https://github.com/behzadanksu/rl-attack"},{"title":"LilTwo/DRL-using-PyTorch","url":"https://github.com/LilTwo/DRL-using-PyTorch"},{"title":"hw9603/DQfD-PyTorch","url":"https://github.com/hw9603/DQfD-PyTorch"},{"title":"thomashirtz/noisy-networks","url":"https://github.com/thomashirtz/noisy-networks"},{"title":"mindspore-courses/Rainbow-MindSpore","url":"https://github.com/mindspore-courses/Rainbow-MindSpore"},{"title":"YanSong97/Master-thesis","url":"https://github.com/YanSong97/Master-thesis"},{"title":"behzadanksu/rlattack-dev","url":"https://github.com/behzadanksu/rlattack-dev"},{"title":"MOVzeroOne/DQN","url":"https://github.com/MOVzeroOne/DQN"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-57","task":"Atari Games","dataset_variant":"Atari-57","rows":11,"metrics":["Mean Human Normalized Score","Human World Record Breakthrough"],"first_row_in_archive_order":{"model":"LBC","paper":"/paper/learnable-behavior-control-breaking-atari","metrics":{"Human World Record Breakthrough":"24","Mean Human Normalized Score":"10077.52%"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-elevator-action","task":"Atari Games","dataset_variant":"Atari 2600 Elevator Action","rows":3,"metrics":["Score"],"first_row_in_archive_order":{"model":"Persistent AL","paper":"/paper/increasing-the-action-gap-new-operators-for","metrics":{"Score":"29100"},"code_links":[{"title":"janhuenermann/neurojs","url":"https://github.com/janhuenermann/neurojs"},{"title":"chainer/chainerrl","url":"https://github.com/chainer/chainerrl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-pooyan","task":"Atari Games","dataset_variant":"Atari 2600 Pooyan","rows":3,"metrics":["Score"],"first_row_in_archive_order":{"model":"UCT","paper":"/paper/the-arcade-learning-environment-an-evaluation","metrics":{"Score":"17763.4"},"code_links":[{"title":"mgbellemare/Arcade-Learning-Environment","url":"https://github.com/mgbellemare/Arcade-Learning-Environment"},{"title":"kenjyoung/MinAtar","url":"https://github.com/kenjyoung/MinAtar"},{"title":"nandomp/AICollaboratory","url":"https://github.com/nandomp/AICollaboratory"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/montezuma-s-revenge-on-atari-2600-montezuma-s","task":"Montezuma's Revenge","dataset_variant":"Atari 2600 Montezuma's Revenge","rows":3,"metrics":["Average Return (NoOp)"],"first_row_in_archive_order":{"model":"Flare","paper":"/paper/reinforcement-learning-with-latent-flow-1","metrics":{"Average Return (NoOp)":"1668"},"code_links":[{"title":"WendyShang/flare","url":"https://github.com/WendyShang/flare"},{"title":"WendyShang/dqn_zoo","url":"https://github.com/WendyShang/dqn_zoo"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-carnival","task":"Atari Games","dataset_variant":"Atari 2600 Carnival","rows":1,"metrics":["Score"],"first_row_in_archive_order":{"model":"UCT","paper":"/paper/the-arcade-learning-environment-an-evaluation","metrics":{"Score":"5132.0"},"code_links":[{"title":"mgbellemare/Arcade-Learning-Environment","url":"https://github.com/mgbellemare/Arcade-Learning-Environment"},{"title":"kenjyoung/MinAtar","url":"https://github.com/kenjyoung/MinAtar"},{"title":"nandomp/AICollaboratory","url":"https://github.com/nandomp/AICollaboratory"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atari-games-on-atari-2600-journey-escape","task":"Atari Games","dataset_variant":"Atari 2600 Journey Escape","rows":1,"metrics":["Score"],"first_row_in_archive_order":{"model":"UCT","paper":"/paper/the-arcade-learning-environment-an-evaluation","metrics":{"Score":"7683.3"},"code_links":[{"title":"mgbellemare/Arcade-Learning-Environment","url":"https://github.com/mgbellemare/Arcade-Learning-Environment"},{"title":"kenjyoung/MinAtar","url":"https://github.com/kenjyoung/MinAtar"},{"title":"nandomp/AICollaboratory","url":"https://github.com/nandomp/AICollaboratory"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/learnable-behavior-control-breaking-atari","title":"Learnable Behavior Control: Breaking Atari Human World Records via Sample-Efficient Behavior Selection","date":"2023-05-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/train-a-real-world-local-path-planner-in-one","title":"Train a Real-world Local Path Planner in One Hour via Partially Decoupled Reinforcement Learning and Vectorized Diversity","date":"2023-05-07","rows_on_this_dataset":51,"code_links":1,"syntology":null},{"paper":"/paper/exploration-by-self-supervised-exploitation","title":"Self-supervised network distillation: an effective approach to exploration in sparse reward environments","date":"2023-02-22","rows_on_this_dataset":14,"code_links":2,"syntology":null},{"paper":"/paper/dna-proximal-policy-optimization-with-a-dual","title":"DNA: Proximal Policy Optimization with a Dual Network Architecture","date":"2022-06-20","rows_on_this_dataset":51,"code_links":1,"syntology":null},{"paper":"/paper/generalized-data-distribution-iteration","title":"Generalized Data Distribution Iteration","date":"2022-06-07","rows_on_this_dataset":114,"code_links":0,"syntology":null},{"paper":"/paper/gdi-rethinking-what-makes-reinforcement-1","title":"GDI: Rethinking What Makes Reinforcement Learning Different from Supervised Learning","date":"2021-11-24","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/iq-learn-inverse-soft-q-learning-for","title":"IQ-Learn: Inverse soft-Q Learning for Imitation","date":"2021-06-23","rows_on_this_dataset":4,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gdi-rethinking-what-makes-reinforcement","title":"GDI: Rethinking What Makes Reinforcement Learning Different From Supervised Learning","date":"2021-06-11","rows_on_this_dataset":36,"code_links":0,"syntology":null},{"paper":"/paper/decision-transformer-reinforcement-learning","title":"Decision Transformer: Reinforcement Learning via Sequence Modeling","date":"2021-06-02","rows_on_this_dataset":4,"code_links":20,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":26,"samples_ran":17,"samples_unverified":9,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/online-and-offline-reinforcement-learning-by","title":"Online and Offline Reinforcement Learning by Planning with a Learned Model","date":"2021-04-13","rows_on_this_dataset":51,"code_links":2,"syntology":null},{"paper":"/paper/improving-computational-efficiency-in-visual","title":"Improving Computational Efficiency in Visual Reinforcement Learning via Stored Embeddings","date":"2021-03-04","rows_on_this_dataset":8,"code_links":1,"syntology":null},{"paper":"/paper/recurrent-rational-networks","title":"Adaptive Rational Activations to Boost Deep Reinforcement Learning","date":"2021-02-18","rows_on_this_dataset":26,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/reinforcement-learning-with-latent-flow-1","title":"Reinforcement Learning with Latent Flow","date":"2021-01-06","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/optimizing-the-neural-architecture-of","title":"Optimizing the Neural Architecture of Reinforcement Learning Agents","date":"2020-11-30","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/discrete-latent-space-world-models-for","title":"Smaller World Models for Reinforcement Learning","date":"2020-10-12","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/mastering-atari-with-discrete-world-models-1","title":"Mastering Atari with Discrete World Models","date":"2020-10-05","rows_on_this_dataset":50,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":3,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/model-free-episodic-control-with-state","title":"Model-Free Episodic Control with State Aggregation","date":"2020-08-21","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/munchausen-reinforcement-learning","title":"Munchausen Reinforcement Learning","date":"2020-07-28","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":22,"samples_ran":12,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/first-return-then-explore","title":"First return, then explore","date":"2020-04-27","rows_on_this_dataset":10,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/curl-contrastive-unsupervised-representations","title":"CURL: Contrastive Unsupervised Representations for Reinforcement Learning","date":"2020-04-08","rows_on_this_dataset":24,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/agent57-outperforming-the-atari-human","title":"Agent57: Outperforming the Atari Human Benchmark","date":"2020-03-30","rows_on_this_dataset":51,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mastering-atari-go-chess-and-shogi-by","title":"Mastering Atari, Go, Chess and Shogi by Planning with a Learned Model","date":"2019-11-19","rows_on_this_dataset":51,"code_links":18,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":64,"samples_ran":43,"samples_unverified":21,"pointer_only_for_licence":62,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fully-parameterized-quantile-function-for","title":"Fully Parameterized Quantile Function for Distributional Reinforcement Learning","date":"2019-11-05","rows_on_this_dataset":23,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":1,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/soft-actor-critic-for-discrete-action","title":"Soft Actor-Critic for Discrete Action Settings","date":"2019-10-16","rows_on_this_dataset":18,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":4,"samples_unverified":14,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/off-policy-actor-critic-with-shared-1","title":"Off-Policy Actor-Critic with Shared Experience Replay","date":"2019-09-25","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/recurrent-independent-mechanisms","title":"Recurrent Independent Mechanisms","date":"2019-09-24","rows_on_this_dataset":3,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":14,"samples_unverified":6,"pointer_only_for_licence":20,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/recurrent-experience-replay-in-distributed","title":"Recurrent Experience Replay in Distributed Reinforcement Learning","date":"2019-05-01","rows_on_this_dataset":52,"code_links":3,"syntology":null},{"paper":"/paper/go-explore-a-new-approach-for-hard","title":"Go-Explore: a New Approach for Hard-Exploration Problems","date":"2019-01-30","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/contingency-aware-exploration-in","title":"Contingency-Aware Exploration in Reinforcement Learning","date":"2018-11-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/exploration-by-random-network-distillation","title":"Exploration by Random Network Distillation","date":"2018-10-30","rows_on_this_dataset":5,"code_links":22,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":43,"samples_ran":26,"samples_unverified":17,"pointer_only_for_licence":15,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/large-scale-study-of-curiosity-driven","title":"Large-Scale Study of Curiosity-Driven Learning","date":"2018-08-13","rows_on_this_dataset":5,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/count-based-exploration-with-the-successor","title":"Count-Based Exploration with the Successor Representation","date":"2018-07-31","rows_on_this_dataset":6,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/policy-optimization-with-penalized-point","title":"Policy Optimization With Penalized Point Probability Distance: An Alternative To Proximal Policy Optimization","date":"2018-07-02","rows_on_this_dataset":45,"code_links":2,"syntology":null},{"paper":"/paper/rudder-return-decomposition-for-delayed","title":"RUDDER: Return Decomposition for Delayed Rewards","date":"2018-06-20","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":3,"samples_unverified":6,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-imitation-learning","title":"Self-Imitation Learning","date":"2018-06-14","rows_on_this_dataset":45,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":1,"samples_unverified":5,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/implicit-quantile-networks-for-distributional","title":"Implicit Quantile Networks for Distributional Reinforcement Learning","date":"2018-06-14","rows_on_this_dataset":51,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/evolving-simple-programs-for-playing-atari","title":"Evolving simple programs for playing Atari games","date":"2018-06-14","rows_on_this_dataset":50,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/playing-atari-with-six-neurons","title":"Playing Atari with Six Neurons","date":"2018-06-04","rows_on_this_dataset":10,"code_links":1,"syntology":null},{"paper":"/paper/distributed-prioritized-experience-replay","title":"Distributed Prioritized Experience Replay","date":"2018-03-02","rows_on_this_dataset":51,"code_links":15,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":0,"samples_unverified":15,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/impala-scalable-distributed-deep-rl-with","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","date":"2018-02-05","rows_on_this_dataset":52,"code_links":24,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":34,"samples_ran":16,"samples_unverified":18,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/distributed-deep-reinforcement-learning-learn","title":"Distributed Deep Reinforcement Learning: Learn how to play Atari games in 21 minutes","date":"2018-01-09","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/distributional-reinforcement-learning-with-1","title":"Distributional Reinforcement Learning with Quantile Regression","date":"2017-10-27","rows_on_this_dataset":51,"code_links":17,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rainbow-combining-improvements-in-deep","title":"Rainbow: Combining Improvements in Deep Reinforcement Learning","date":"2017-10-06","rows_on_this_dataset":4,"code_links":34,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mean-actor-critic","title":"Mean Actor Critic","date":"2017-09-01","rows_on_this_dataset":6,"code_links":2,"syntology":null},{"paper":"/paper/a-distributional-perspective-on-reinforcement","title":"A Distributional Perspective on Reinforcement Learning","date":"2017-07-21","rows_on_this_dataset":45,"code_links":22,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/value-prediction-network","title":"Value Prediction Network","date":"2017-07-11","rows_on_this_dataset":8,"code_links":2,"syntology":null},{"paper":"/paper/noisy-networks-for-exploration","title":"Noisy Networks for Exploration","date":"2017-06-30","rows_on_this_dataset":48,"code_links":15,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/count-based-exploration-in-feature-space-for","title":"Count-Based Exploration in Feature Space for Reinforcement Learning","date":"2017-06-25","rows_on_this_dataset":10,"code_links":1,"syntology":null},{"paper":"/paper/the-reactor-a-fast-and-sample-efficient-actor","title":"The Reactor: A fast and sample-efficient Actor-Critic agent for Reinforcement Learning","date":"2017-04-15","rows_on_this_dataset":16,"code_links":0,"syntology":null},{"paper":"/paper/evolution-strategies-as-a-scalable","title":"Evolution Strategies as a Scalable Alternative to Reinforcement Learning","date":"2017-03-10","rows_on_this_dataset":41,"code_links":23,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":29,"samples_ran":7,"samples_unverified":22,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/count-based-exploration-with-neural-density","title":"Count-Based Exploration with Neural Density Models","date":"2017-03-03","rows_on_this_dataset":9,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exploration-a-study-of-count-based","title":"#Exploration: A Study of Count-Based Exploration for Deep Reinforcement Learning","date":"2016-11-15","rows_on_this_dataset":4,"code_links":3,"syntology":null},{"paper":"/paper/unifying-count-based-exploration-and","title":"Unifying Count-Based Exploration and Intrinsic Motivation","date":"2016-06-06","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/learning-values-across-many-orders-of","title":"Learning values across many orders of magnitude","date":"2016-02-24","rows_on_this_dataset":45,"code_links":0,"syntology":null},{"paper":"/paper/deep-exploration-via-bootstrapped-dqn","title":"Deep Exploration via Bootstrapped DQN","date":"2016-02-15","rows_on_this_dataset":45,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/asynchronous-methods-for-deep-reinforcement","title":"Asynchronous Methods for Deep Reinforcement Learning","date":"2016-02-04","rows_on_this_dataset":138,"code_links":70,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":95,"samples_ran":38,"samples_unverified":57,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/increasing-the-action-gap-new-operators-for","title":"Increasing the Action Gap: New Operators for Reinforcement Learning","date":"2015-12-15","rows_on_this_dataset":88,"code_links":2,"syntology":null},{"paper":"/paper/deep-attention-recurrent-q-network","title":"Deep Attention Recurrent Q-Network","date":"2015-12-05","rows_on_this_dataset":5,"code_links":3,"syntology":null},{"paper":"/paper/dueling-network-architectures-for-deep","title":"Dueling Network Architectures for Deep Reinforcement Learning","date":"2015-11-20","rows_on_this_dataset":185,"code_links":73,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/prioritized-experience-replay","title":"Prioritized Experience Replay","date":"2015-11-18","rows_on_this_dataset":91,"code_links":77,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":111,"samples_ran":78,"samples_unverified":33,"pointer_only_for_licence":43,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deep-reinforcement-learning-with-double-q","title":"Deep Reinforcement Learning with Double Q-learning","date":"2015-09-22","rows_on_this_dataset":183,"code_links":97,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":106,"samples_ran":55,"samples_unverified":51,"pointer_only_for_licence":57,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/massively-parallel-methods-for-deep","title":"Massively Parallel Methods for Deep Reinforcement Learning","date":"2015-07-15","rows_on_this_dataset":45,"code_links":3,"syntology":null},{"paper":"/paper/incentivizing-exploration-in-reinforcement","title":"Incentivizing Exploration In Reinforcement Learning With Deep Predictive Models","date":"2015-07-03","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/human-level-control-through-deep","title":"Human level control through deep reinforcement learning","date":"2015-02-25","rows_on_this_dataset":45,"code_links":8,"syntology":null},{"paper":"/paper/playing-atari-with-deep-reinforcement","title":"Playing Atari with Deep Reinforcement Learning","date":"2013-12-19","rows_on_this_dataset":6,"code_links":112,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":117,"samples_ran":56,"samples_unverified":61,"pointer_only_for_licence":56,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/the-arcade-learning-environment-an-evaluation","title":"The Arcade Learning Environment: An Evaluation Platform for General Agents","date":"2012-07-19","rows_on_this_dataset":98,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":36,"samples_harvested":819,"samples_ran":410,"samples_unverified":409,"pointer_only_for_licence":323,"papers_with_no_sample_that_ran":6,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}