{"url":"/dataset/openai-gym","name":"OpenAI Gym","full_name":"OpenAI Gym","description_markdown":"**OpenAI Gym** is a toolkit for developing and comparing reinforcement learning algorithms. It includes environment such as Algorithmic, Atari, Box2D, Classic Control, MuJoCo, Robotics, and Toy Text.\r\n\r\nSource: [https://github.com/openai/gym](https://github.com/openai/gym)\r\nImage Source: [https://medium.com/@tuzzer/cart-pole-balancing-with-q-learning-b54c6068d947](https://medium.com/@tuzzer/cart-pole-balancing-with-q-learning-b54c6068d947)","description_withheld":null,"homepage":"https://gym.openai.com/","introduced_date":"2016-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/openai-gym","title":"OpenAI Gym","first_author":"Greg Brockman","url":null},"license":{"name":"MIT","url":"https://github.com/openai/gym/blob/master/LICENSE.md"},"modalities":[{"name":"Environment","url":"/datasets/modality/environment"}],"tasks":[{"name":"Continuous Control","url":"/task/continuous-control","datasets_with_task":"/datasets/task/continuous-control"},{"name":"OpenAI Gym","url":"/task/openai-gym","datasets_with_task":"/datasets/task/openai-gym"}],"languages":[{"name":"Azerbaijani","url":"/datasets/language/azerbaijani"}],"variants":["OpenAI Gym","Cart Pole (OpenAI Gym)","Lunar Lander (OpenAI Gym)","Pendulum-v1"],"data_loaders":[{"repo":"https://github.com/openai/gym","url":"https://github.com/openai/gym/blob/master/docs/environments.md","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":1305,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/continuous-control-on-lunar-lander-openai-gym","task":"Continuous Control","dataset_variant":"Lunar Lander (OpenAI Gym)","rows":5,"metrics":["Score"],"first_row_in_archive_order":{"model":"SAC","paper":"/paper/soft-actor-critic-off-policy-maximum-entropy","metrics":{"Score":"284.59±0.97"},"code_links":[{"title":"ray-project/ray","url":"https://github.com/ray-project/ray/tree/master/rllib"},{"title":"DLR-RM/stable-baselines3","url":"https://github.com/DLR-RM/stable-baselines3"},{"title":"hill-a/stable-baselines","url":"https://github.com/hill-a/stable-baselines"},{"title":"facebookresearch/ReAgent","url":"https://github.com/facebookresearch/ReAgent"},{"title":"opendilab/DI-engine","url":"https://github.com/opendilab/DI-engine/blob/main/ding/policy/sac.py"},{"title":"pytorch/rl","url":"https://github.com/pytorch/rl/tree/main/examples/sac"},{"title":"facebookresearch/rl","url":"https://github.com/facebookresearch/rl/blob/main/examples/sac/sac.py"},{"title":"quantumiracle/Popular-RL-Algorithms","url":"https://github.com/quantumiracle/Popular-RL-Algorithms"},{"title":"haarnoja/sac","url":"https://github.com/haarnoja/sac"},{"title":"MrSyee/pg-is-all-you-need","url":"https://github.com/MrSyee/pg-is-all-you-need"},{"title":"Rafael1s/Deep-Reinforcement-Learning-Udacity","url":"https://github.com/Rafael1s/Deep-Reinforcement-Learning-Udacity"},{"title":"pranz24/pytorch-soft-actor-critic","url":"https://github.com/pranz24/pytorch-soft-actor-critic"},{"title":"toni-sm/skrl","url":"https://github.com/toni-sm/skrl"},{"title":"ikostrikov/jax-rl","url":"https://github.com/ikostrikov/jax-rl"},{"title":"tensorlayer/RLzoo","url":"https://github.com/tensorlayer/RLzoo"},{"title":"marload/DeepRL-TensorFlow2","url":"https://github.com/marload/DeepRL-TensorFlow2"},{"title":"trackmania-rl/tmrl","url":"https://github.com/trackmania-rl/tmrl"},{"title":"Kaixhin/imitation-learning","url":"https://github.com/Kaixhin/imitation-learning"},{"title":"araffin/sbx","url":"https://github.com/araffin/sbx"},{"title":"andrejorsula/drl_grasping","url":"https://github.com/andrejorsula/drl_grasping"},{"title":"BY571/Soft-Actor-Critic-and-Extensions","url":"https://github.com/BY571/Soft-Actor-Critic-and-Extensions"},{"title":"ShawK91/Evolutionary-Reinforcement-Learning","url":"https://github.com/ShawK91/Evolutionary-Reinforcement-Learning"},{"title":"ShawK91/erl_paper_nips18","url":"https://github.com/ShawK91/erl_paper_nips18"},{"title":"ku2482/gail-airl-ppo.pytorch","url":"https://github.com/ku2482/gail-airl-ppo.pytorch"},{"title":"learn-to-race/l2r","url":"https://github.com/learn-to-race/l2r"},{"title":"dasgringuen/assetto_corsa_gym","url":"https://github.com/dasgringuen/assetto_corsa_gym"},{"title":"polixir/NeoRL","url":"https://github.com/polixir/NeoRL"},{"title":"toshikwa/soft-actor-critic.pytorch","url":"https://github.com/toshikwa/soft-actor-critic.pytorch"},{"title":"ku2482/soft-actor-critic.pytorch","url":"https://github.com/ku2482/soft-actor-critic.pytorch"},{"title":"ku2482/rljax","url":"https://github.com/ku2482/rljax"},{"title":"jakegrigsby/deep_control","url":"https://github.com/jakegrigsby/deep_control/blob/master/deep_control/sac.py"},{"title":"ac-93/soft-actor-critic","url":"https://github.com/ac-93/soft-actor-critic"},{"title":"RLAgent/state-marginal-matching","url":"https://github.com/RLAgent/state-marginal-matching"},{"title":"dfki-ric-underactuated-lab/torque_limited_simple_pendulum","url":"https://github.com/dfki-ric-underactuated-lab/torque_limited_simple_pendulum"},{"title":"FOCAL-ICLR/FOCAL-ICLR","url":"https://github.com/FOCAL-ICLR/FOCAL-ICLR"},{"title":"lanqingli1993/focal-iclr","url":"https://github.com/lanqingli1993/focal-iclr"},{"title":"toshikwa/discor.pytorch","url":"https://github.com/toshikwa/discor.pytorch"},{"title":"ku2482/discor.pytorch","url":"https://github.com/ku2482/discor.pytorch"},{"title":"tilkb/thermoai","url":"https://github.com/tilkb/thermoai"},{"title":"kdally/fault-tolerant-flight-control-drl","url":"https://github.com/kdally/fault-tolerant-flight-control-drl"},{"title":"kairproject/kair_algorithms_draft","url":"https://github.com/kairproject/kair_algorithms_draft"},{"title":"fdcl-gwu/gym-rotor","url":"https://github.com/fdcl-gwu/gym-rotor"},{"title":"roythuly/obac","url":"https://github.com/roythuly/obac"},{"title":"kushagra06/SAC","url":"https://github.com/kushagra06/SAC"},{"title":"yining043/SAC-discrete","url":"https://github.com/yining043/SAC-discrete"},{"title":"MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning","url":"https://github.com/MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning"},{"title":"AutumnWu/Streamlined-Off-Policy-Learning","url":"https://github.com/AutumnWu/Streamlined-Off-Policy-Learning"},{"title":"core-robotics-lab/icct","url":"https://github.com/core-robotics-lab/icct"},{"title":"xiuyu0000/new_papers_codes","url":"https://github.com/xiuyu0000/new_papers_codes/tree/main/sac"},{"title":"ku2482/rltorch","url":"https://github.com/ku2482/rltorch"},{"title":"baturaysaglam/la3p","url":"https://github.com/baturaysaglam/la3p"},{"title":"seungju-k1m/sac-td3-td7","url":"https://github.com/seungju-k1m/sac-td3-td7"},{"title":"X3N4/car_racer","url":"https://github.com/X3N4/car_racer"},{"title":"timoklein/car_racer","url":"https://github.com/timoklein/car_racer"},{"title":"moreanp/csro","url":"https://github.com/moreanp/csro"},{"title":"ajaysub110/rl-pytorch","url":"https://github.com/ajaysub110/rl-pytorch"},{"title":"watchernyu/spinningup-drl-prototyping","url":"https://github.com/watchernyu/spinningup-drl-prototyping"},{"title":"tliu1997/rnac","url":"https://github.com/tliu1997/rnac"},{"title":"thomashirtz/pytorch-soft-actor-critic","url":"https://github.com/thomashirtz/pytorch-soft-actor-critic"},{"title":"thomashirtz/soft-actor-critic","url":"https://github.com/thomashirtz/soft-actor-critic"},{"title":"ccolas/rl_stats","url":"https://github.com/ccolas/rl_stats"},{"title":"nagisazj/idaq_public","url":"https://github.com/nagisazj/idaq_public"},{"title":"Ipsedo/EvoMotion","url":"https://github.com/Ipsedo/EvoMotion"},{"title":"lucadellalib/sac-beta","url":"https://github.com/lucadellalib/sac-beta"},{"title":"h-aboutalebi/SparceReward","url":"https://github.com/h-aboutalebi/SparceReward"},{"title":"garyzyr001/rethinking-airl","url":"https://github.com/garyzyr001/rethinking-airl"},{"title":"flowersteam/rl_stats","url":"https://github.com/flowersteam/rl_stats"},{"title":"gijskoning/ReproducingCURL","url":"https://github.com/gijskoning/ReproducingCURL"},{"title":"yimingpeng/sac-master","url":"https://github.com/yimingpeng/sac-master"},{"title":"MarsEleven/car_racer_RL","url":"https://github.com/MarsEleven/car_racer_RL"},{"title":"yhisaki/average-reward-drl","url":"https://github.com/yhisaki/average-reward-drl"},{"title":"AmmarFayad/Behavioral-Actor-Critic","url":"https://github.com/AmmarFayad/Behavioral-Actor-Critic"},{"title":"autumnwu/aggressive-q-learning-with-ensembles","url":"https://github.com/autumnwu/aggressive-q-learning-with-ensembles"},{"title":"tarod13/SAC","url":"https://github.com/tarod13/SAC"},{"title":"Steinheilig/Imbiss","url":"https://github.com/Steinheilig/Imbiss"},{"title":"tmjeong1103/RL_with_RAY","url":"https://github.com/tmjeong1103/RL_with_RAY"},{"title":"QuentinVacher-rl/SoftActorCritic-in-Cpp-using-LibTorch","url":"https://github.com/QuentinVacher-rl/SoftActorCritic-in-Cpp-using-LibTorch"},{"title":"SaminYeasar/off_policy_ac","url":"https://github.com/SaminYeasar/off_policy_ac"},{"title":"lollcat/Soft-Actor-Critic","url":"https://github.com/lollcat/Soft-Actor-Critic"},{"title":"donamin/llc","url":"https://github.com/donamin/llc"},{"title":"susan-amin/SparseBaseline1","url":"https://github.com/susan-amin/SparseBaseline1"},{"title":"hyunin-lee/ForecasterSAC","url":"https://github.com/hyunin-lee/ForecasterSAC"},{"title":"cindycia/Atari-SAC-Discrete","url":"https://github.com/cindycia/Atari-SAC-Discrete"},{"title":"mxblr/DeepRLHockey","url":"https://github.com/mxblr/DeepRLHockey"},{"title":"rk1998/robot-sac","url":"https://github.com/rk1998/robot-sac"},{"title":"sunfex/weighted-sac","url":"https://github.com/sunfex/weighted-sac"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/openai-gym-on-pendulum-v1","task":"OpenAI Gym","dataset_variant":"Pendulum-v1","rows":2,"metrics":["Mean Reward","Average Decisions","Action Repetition"],"first_row_in_archive_order":{"model":"TLA with Hierarchical Reward Functions","paper":"/paper/creating-hierarchical-dispositions-of-needs","metrics":{"Action Repetition":".8073","Average Decisions":"38.6","Mean Reward":"-125.02"},"code_links":[{"title":"TofaraMoyo/Heirachical-Reward-Functions","url":"https://github.com/TofaraMoyo/Heirachical-Reward-Functions"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/continuous-control-on-cart-pole-openai-gym","task":"Continuous Control","dataset_variant":"Cart Pole (OpenAI Gym)","rows":1,"metrics":["Score"],"first_row_in_archive_order":{"model":"MAC","paper":"/paper/mean-actor-critic","metrics":{"Score":"178.3"},"code_links":[{"title":"kavosh8/MAC","url":"https://github.com/kavosh8/MAC"},{"title":"camall3n/atari-MAC","url":"https://github.com/camall3n/atari-MAC"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/creating-hierarchical-dispositions-of-needs","title":"Creating Hierarchical Dispositions of Needs in an Agent","date":"2024-11-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/temporally-layered-architecture-for-efficient","title":"Optimizing Attention and Cognitive Control Costs Using Temporally-Layered Architectures","date":"2023-05-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/addressing-function-approximation-error-in","title":"Addressing Function Approximation Error in Actor-Critic Methods","date":"2018-02-26","rows_on_this_dataset":1,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":36,"samples_ran":9,"samples_unverified":27,"pointer_only_for_licence":20,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/soft-actor-critic-off-policy-maximum-entropy","title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","date":"2018-01-04","rows_on_this_dataset":1,"code_links":86,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":148,"samples_ran":91,"samples_unverified":57,"pointer_only_for_licence":66,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mean-actor-critic","title":"Mean Actor Critic","date":"2017-09-01","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/proximal-policy-optimization-algorithms","title":"Proximal Policy Optimization Algorithms","date":"2017-07-20","rows_on_this_dataset":1,"code_links":188,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":176,"samples_ran":99,"samples_unverified":77,"pointer_only_for_licence":94,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","rows_on_this_dataset":1,"code_links":161,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":306,"samples_ran":158,"samples_unverified":148,"pointer_only_for_licence":163,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":666,"samples_ran":357,"samples_unverified":309,"pointer_only_for_licence":343,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}