{"url":"/task/openai-gym","name":"OpenAI Gym","slug":"openai-gym","description_markdown":"An open-source toolkit from OpenAI that implements several Reinforcement Learning benchmarks including: classic control, Atari, Robotics and MuJoCo tasks.\r\n\r\n(Description by [Evolutionary learning of interpretable decision trees](https://paperswithcode.com/paper/evolutionary-learning-of-interpretable))\r\n\r\n(Image Credit: [OpenAI Gym](https://gym.openai.com/))","categories":[{"name":"Playing Games","url":"/area/playing-games"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":382,"papers_with_code":179,"benchmarks":17,"benchmark_tables_in_archive":17,"benchmark_tables_shown":17,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/openai-gym-on-ant-v4","slug":"openai-gym-on-ant-v4","dataset":"Ant-v4","dataset_url":null,"rows_in_archive":5,"metrics":["Average Return"],"first_row_in_archive_order":{"model":"MEow","paper_title":"Maximum Entropy Reinforcement Learning via Energy-Based Normalizing Flow","paper_url":"/paper/maximum-entropy-reinforcement-learning-via","paper_date":"2024-05-22","arxiv_id":"2405.13629","code_links":[{"title":"ChienFeng-hub/meow","url":"https://github.com/ChienFeng-hub/meow"}],"syntology":null}},{"leaderboard":"/sota/openai-gym-on-halfcheetah-v4","slug":"openai-gym-on-halfcheetah-v4","dataset":"HalfCheetah-v4","dataset_url":null,"rows_in_archive":5,"metrics":["Average Return"],"first_row_in_archive_order":{"model":"SAC","paper_title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","paper_url":"/paper/soft-actor-critic-off-policy-maximum-entropy","paper_date":"2018-01-04","arxiv_id":"1801.01290","code_links":[{"title":"ray-project/ray","url":"https://github.com/ray-project/ray/tree/master/rllib"},{"title":"DLR-RM/stable-baselines3","url":"https://github.com/DLR-RM/stable-baselines3"},{"title":"hill-a/stable-baselines","url":"https://github.com/hill-a/stable-baselines"},{"title":"facebookresearch/ReAgent","url":"https://github.com/facebookresearch/ReAgent"},{"title":"opendilab/DI-engine","url":"https://github.com/opendilab/DI-engine/blob/main/ding/policy/sac.py"},{"title":"pytorch/rl","url":"https://github.com/pytorch/rl/tree/main/examples/sac"},{"title":"facebookresearch/rl","url":"https://github.com/facebookresearch/rl/blob/main/examples/sac/sac.py"},{"title":"quantumiracle/Popular-RL-Algorithms","url":"https://github.com/quantumiracle/Popular-RL-Algorithms"},{"title":"haarnoja/sac","url":"https://github.com/haarnoja/sac"},{"title":"MrSyee/pg-is-all-you-need","url":"https://github.com/MrSyee/pg-is-all-you-need"},{"title":"Rafael1s/Deep-Reinforcement-Learning-Udacity","url":"https://github.com/Rafael1s/Deep-Reinforcement-Learning-Udacity"},{"title":"pranz24/pytorch-soft-actor-critic","url":"https://github.com/pranz24/pytorch-soft-actor-critic"},{"title":"toni-sm/skrl","url":"https://github.com/toni-sm/skrl"},{"title":"ikostrikov/jax-rl","url":"https://github.com/ikostrikov/jax-rl"},{"title":"tensorlayer/RLzoo","url":"https://github.com/tensorlayer/RLzoo"},{"title":"marload/DeepRL-TensorFlow2","url":"https://github.com/marload/DeepRL-TensorFlow2"},{"title":"trackmania-rl/tmrl","url":"https://github.com/trackmania-rl/tmrl"},{"title":"Kaixhin/imitation-learning","url":"https://github.com/Kaixhin/imitation-learning"},{"title":"araffin/sbx","url":"https://github.com/araffin/sbx"},{"title":"andrejorsula/drl_grasping","url":"https://github.com/andrejorsula/drl_grasping"},{"title":"BY571/Soft-Actor-Critic-and-Extensions","url":"https://github.com/BY571/Soft-Actor-Critic-and-Extensions"},{"title":"ShawK91/Evolutionary-Reinforcement-Learning","url":"https://github.com/ShawK91/Evolutionary-Reinforcement-Learning"},{"title":"ShawK91/erl_paper_nips18","url":"https://github.com/ShawK91/erl_paper_nips18"},{"title":"ku2482/gail-airl-ppo.pytorch","url":"https://github.com/ku2482/gail-airl-ppo.pytorch"},{"title":"learn-to-race/l2r","url":"https://github.com/learn-to-race/l2r"},{"title":"dasgringuen/assetto_corsa_gym","url":"https://github.com/dasgringuen/assetto_corsa_gym"},{"title":"polixir/NeoRL","url":"https://github.com/polixir/NeoRL"},{"title":"toshikwa/soft-actor-critic.pytorch","url":"https://github.com/toshikwa/soft-actor-critic.pytorch"},{"title":"ku2482/soft-actor-critic.pytorch","url":"https://github.com/ku2482/soft-actor-critic.pytorch"},{"title":"ku2482/rljax","url":"https://github.com/ku2482/rljax"},{"title":"jakegrigsby/deep_control","url":"https://github.com/jakegrigsby/deep_control/blob/master/deep_control/sac.py"},{"title":"ac-93/soft-actor-critic","url":"https://github.com/ac-93/soft-actor-critic"},{"title":"RLAgent/state-marginal-matching","url":"https://github.com/RLAgent/state-marginal-matching"},{"title":"dfki-ric-underactuated-lab/torque_limited_simple_pendulum","url":"https://github.com/dfki-ric-underactuated-lab/torque_limited_simple_pendulum"},{"title":"FOCAL-ICLR/FOCAL-ICLR","url":"https://github.com/FOCAL-ICLR/FOCAL-ICLR"},{"title":"lanqingli1993/focal-iclr","url":"https://github.com/lanqingli1993/focal-iclr"},{"title":"toshikwa/discor.pytorch","url":"https://github.com/toshikwa/discor.pytorch"},{"title":"ku2482/discor.pytorch","url":"https://github.com/ku2482/discor.pytorch"},{"title":"tilkb/thermoai","url":"https://github.com/tilkb/thermoai"},{"title":"kdally/fault-tolerant-flight-control-drl","url":"https://github.com/kdally/fault-tolerant-flight-control-drl"},{"title":"kairproject/kair_algorithms_draft","url":"https://github.com/kairproject/kair_algorithms_draft"},{"title":"fdcl-gwu/gym-rotor","url":"https://github.com/fdcl-gwu/gym-rotor"},{"title":"roythuly/obac","url":"https://github.com/roythuly/obac"},{"title":"kushagra06/SAC","url":"https://github.com/kushagra06/SAC"},{"title":"yining043/SAC-discrete","url":"https://github.com/yining043/SAC-discrete"},{"title":"MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning","url":"https://github.com/MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning"},{"title":"AutumnWu/Streamlined-Off-Policy-Learning","url":"https://github.com/AutumnWu/Streamlined-Off-Policy-Learning"},{"title":"core-robotics-lab/icct","url":"https://github.com/core-robotics-lab/icct"},{"title":"xiuyu0000/new_papers_codes","url":"https://github.com/xiuyu0000/new_papers_codes/tree/main/sac"},{"title":"ku2482/rltorch","url":"https://github.com/ku2482/rltorch"},{"title":"baturaysaglam/la3p","url":"https://github.com/baturaysaglam/la3p"},{"title":"seungju-k1m/sac-td3-td7","url":"https://github.com/seungju-k1m/sac-td3-td7"},{"title":"X3N4/car_racer","url":"https://github.com/X3N4/car_racer"},{"title":"timoklein/car_racer","url":"https://github.com/timoklein/car_racer"},{"title":"moreanp/csro","url":"https://github.com/moreanp/csro"},{"title":"ajaysub110/rl-pytorch","url":"https://github.com/ajaysub110/rl-pytorch"},{"title":"watchernyu/spinningup-drl-prototyping","url":"https://github.com/watchernyu/spinningup-drl-prototyping"},{"title":"tliu1997/rnac","url":"https://github.com/tliu1997/rnac"},{"title":"thomashirtz/pytorch-soft-actor-critic","url":"https://github.com/thomashirtz/pytorch-soft-actor-critic"},{"title":"thomashirtz/soft-actor-critic","url":"https://github.com/thomashirtz/soft-actor-critic"},{"title":"ccolas/rl_stats","url":"https://github.com/ccolas/rl_stats"},{"title":"nagisazj/idaq_public","url":"https://github.com/nagisazj/idaq_public"},{"title":"Ipsedo/EvoMotion","url":"https://github.com/Ipsedo/EvoMotion"},{"title":"lucadellalib/sac-beta","url":"https://github.com/lucadellalib/sac-beta"},{"title":"h-aboutalebi/SparceReward","url":"https://github.com/h-aboutalebi/SparceReward"},{"title":"garyzyr001/rethinking-airl","url":"https://github.com/garyzyr001/rethinking-airl"},{"title":"flowersteam/rl_stats","url":"https://github.com/flowersteam/rl_stats"},{"title":"gijskoning/ReproducingCURL","url":"https://github.com/gijskoning/ReproducingCURL"},{"title":"yimingpeng/sac-master","url":"https://github.com/yimingpeng/sac-master"},{"title":"MarsEleven/car_racer_RL","url":"https://github.com/MarsEleven/car_racer_RL"},{"title":"yhisaki/average-reward-drl","url":"https://github.com/yhisaki/average-reward-drl"},{"title":"AmmarFayad/Behavioral-Actor-Critic","url":"https://github.com/AmmarFayad/Behavioral-Actor-Critic"},{"title":"autumnwu/aggressive-q-learning-with-ensembles","url":"https://github.com/autumnwu/aggressive-q-learning-with-ensembles"},{"title":"tarod13/SAC","url":"https://github.com/tarod13/SAC"},{"title":"Steinheilig/Imbiss","url":"https://github.com/Steinheilig/Imbiss"},{"title":"tmjeong1103/RL_with_RAY","url":"https://github.com/tmjeong1103/RL_with_RAY"},{"title":"QuentinVacher-rl/SoftActorCritic-in-Cpp-using-LibTorch","url":"https://github.com/QuentinVacher-rl/SoftActorCritic-in-Cpp-using-LibTorch"},{"title":"SaminYeasar/off_policy_ac","url":"https://github.com/SaminYeasar/off_policy_ac"},{"title":"lollcat/Soft-Actor-Critic","url":"https://github.com/lollcat/Soft-Actor-Critic"},{"title":"donamin/llc","url":"https://github.com/donamin/llc"},{"title":"susan-amin/SparseBaseline1","url":"https://github.com/susan-amin/SparseBaseline1"},{"title":"hyunin-lee/ForecasterSAC","url":"https://github.com/hyunin-lee/ForecasterSAC"},{"title":"cindycia/Atari-SAC-Discrete","url":"https://github.com/cindycia/Atari-SAC-Discrete"},{"title":"mxblr/DeepRLHockey","url":"https://github.com/mxblr/DeepRLHockey"},{"title":"rk1998/robot-sac","url":"https://github.com/rk1998/robot-sac"},{"title":"sunfex/weighted-sac","url":"https://github.com/sunfex/weighted-sac"}],"syntology":{"n":148,"n_ran":91,"n_unverified":57,"n_pointer_only":66}}},{"leaderboard":"/sota/openai-gym-on-hopper-v4","slug":"openai-gym-on-hopper-v4","dataset":"Hopper-v4","dataset_url":null,"rows_in_archive":5,"metrics":["Average Return"],"first_row_in_archive_order":{"model":"MEow","paper_title":"Maximum Entropy Reinforcement Learning via Energy-Based Normalizing Flow","paper_url":"/paper/maximum-entropy-reinforcement-learning-via","paper_date":"2024-05-22","arxiv_id":"2405.13629","code_links":[{"title":"ChienFeng-hub/meow","url":"https://github.com/ChienFeng-hub/meow"}],"syntology":null}},{"leaderboard":"/sota/openai-gym-on-humanoid-v4","slug":"openai-gym-on-humanoid-v4","dataset":"Humanoid-v4","dataset_url":null,"rows_in_archive":5,"metrics":["Average Return"],"first_row_in_archive_order":{"model":"MEow","paper_title":"Maximum Entropy Reinforcement Learning via Energy-Based Normalizing Flow","paper_url":"/paper/maximum-entropy-reinforcement-learning-via","paper_date":"2024-05-22","arxiv_id":"2405.13629","code_links":[{"title":"ChienFeng-hub/meow","url":"https://github.com/ChienFeng-hub/meow"}],"syntology":null}},{"leaderboard":"/sota/openai-gym-on-walker2d-v4","slug":"openai-gym-on-walker2d-v4","dataset":"Walker2d-v4","dataset_url":null,"rows_in_archive":5,"metrics":["Average Return"],"first_row_in_archive_order":{"model":"SAC","paper_title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","paper_url":"/paper/soft-actor-critic-off-policy-maximum-entropy","paper_date":"2018-01-04","arxiv_id":"1801.01290","code_links":[{"title":"ray-project/ray","url":"https://github.com/ray-project/ray/tree/master/rllib"},{"title":"DLR-RM/stable-baselines3","url":"https://github.com/DLR-RM/stable-baselines3"},{"title":"hill-a/stable-baselines","url":"https://github.com/hill-a/stable-baselines"},{"title":"facebookresearch/ReAgent","url":"https://github.com/facebookresearch/ReAgent"},{"title":"opendilab/DI-engine","url":"https://github.com/opendilab/DI-engine/blob/main/ding/policy/sac.py"},{"title":"pytorch/rl","url":"https://github.com/pytorch/rl/tree/main/examples/sac"},{"title":"facebookresearch/rl","url":"https://github.com/facebookresearch/rl/blob/main/examples/sac/sac.py"},{"title":"quantumiracle/Popular-RL-Algorithms","url":"https://github.com/quantumiracle/Popular-RL-Algorithms"},{"title":"haarnoja/sac","url":"https://github.com/haarnoja/sac"},{"title":"MrSyee/pg-is-all-you-need","url":"https://github.com/MrSyee/pg-is-all-you-need"},{"title":"Rafael1s/Deep-Reinforcement-Learning-Udacity","url":"https://github.com/Rafael1s/Deep-Reinforcement-Learning-Udacity"},{"title":"pranz24/pytorch-soft-actor-critic","url":"https://github.com/pranz24/pytorch-soft-actor-critic"},{"title":"toni-sm/skrl","url":"https://github.com/toni-sm/skrl"},{"title":"ikostrikov/jax-rl","url":"https://github.com/ikostrikov/jax-rl"},{"title":"tensorlayer/RLzoo","url":"https://github.com/tensorlayer/RLzoo"},{"title":"marload/DeepRL-TensorFlow2","url":"https://github.com/marload/DeepRL-TensorFlow2"},{"title":"trackmania-rl/tmrl","url":"https://github.com/trackmania-rl/tmrl"},{"title":"Kaixhin/imitation-learning","url":"https://github.com/Kaixhin/imitation-learning"},{"title":"araffin/sbx","url":"https://github.com/araffin/sbx"},{"title":"andrejorsula/drl_grasping","url":"https://github.com/andrejorsula/drl_grasping"},{"title":"BY571/Soft-Actor-Critic-and-Extensions","url":"https://github.com/BY571/Soft-Actor-Critic-and-Extensions"},{"title":"ShawK91/Evolutionary-Reinforcement-Learning","url":"https://github.com/ShawK91/Evolutionary-Reinforcement-Learning"},{"title":"ShawK91/erl_paper_nips18","url":"https://github.com/ShawK91/erl_paper_nips18"},{"title":"ku2482/gail-airl-ppo.pytorch","url":"https://github.com/ku2482/gail-airl-ppo.pytorch"},{"title":"learn-to-race/l2r","url":"https://github.com/learn-to-race/l2r"},{"title":"dasgringuen/assetto_corsa_gym","url":"https://github.com/dasgringuen/assetto_corsa_gym"},{"title":"polixir/NeoRL","url":"https://github.com/polixir/NeoRL"},{"title":"toshikwa/soft-actor-critic.pytorch","url":"https://github.com/toshikwa/soft-actor-critic.pytorch"},{"title":"ku2482/soft-actor-critic.pytorch","url":"https://github.com/ku2482/soft-actor-critic.pytorch"},{"title":"ku2482/rljax","url":"https://github.com/ku2482/rljax"},{"title":"jakegrigsby/deep_control","url":"https://github.com/jakegrigsby/deep_control/blob/master/deep_control/sac.py"},{"title":"ac-93/soft-actor-critic","url":"https://github.com/ac-93/soft-actor-critic"},{"title":"RLAgent/state-marginal-matching","url":"https://github.com/RLAgent/state-marginal-matching"},{"title":"dfki-ric-underactuated-lab/torque_limited_simple_pendulum","url":"https://github.com/dfki-ric-underactuated-lab/torque_limited_simple_pendulum"},{"title":"FOCAL-ICLR/FOCAL-ICLR","url":"https://github.com/FOCAL-ICLR/FOCAL-ICLR"},{"title":"lanqingli1993/focal-iclr","url":"https://github.com/lanqingli1993/focal-iclr"},{"title":"toshikwa/discor.pytorch","url":"https://github.com/toshikwa/discor.pytorch"},{"title":"ku2482/discor.pytorch","url":"https://github.com/ku2482/discor.pytorch"},{"title":"tilkb/thermoai","url":"https://github.com/tilkb/thermoai"},{"title":"kdally/fault-tolerant-flight-control-drl","url":"https://github.com/kdally/fault-tolerant-flight-control-drl"},{"title":"kairproject/kair_algorithms_draft","url":"https://github.com/kairproject/kair_algorithms_draft"},{"title":"fdcl-gwu/gym-rotor","url":"https://github.com/fdcl-gwu/gym-rotor"},{"title":"roythuly/obac","url":"https://github.com/roythuly/obac"},{"title":"kushagra06/SAC","url":"https://github.com/kushagra06/SAC"},{"title":"yining043/SAC-discrete","url":"https://github.com/yining043/SAC-discrete"},{"title":"MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning","url":"https://github.com/MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning"},{"title":"AutumnWu/Streamlined-Off-Policy-Learning","url":"https://github.com/AutumnWu/Streamlined-Off-Policy-Learning"},{"title":"core-robotics-lab/icct","url":"https://github.com/core-robotics-lab/icct"},{"title":"xiuyu0000/new_papers_codes","url":"https://github.com/xiuyu0000/new_papers_codes/tree/main/sac"},{"title":"ku2482/rltorch","url":"https://github.com/ku2482/rltorch"},{"title":"baturaysaglam/la3p","url":"https://github.com/baturaysaglam/la3p"},{"title":"seungju-k1m/sac-td3-td7","url":"https://github.com/seungju-k1m/sac-td3-td7"},{"title":"X3N4/car_racer","url":"https://github.com/X3N4/car_racer"},{"title":"timoklein/car_racer","url":"https://github.com/timoklein/car_racer"},{"title":"moreanp/csro","url":"https://github.com/moreanp/csro"},{"title":"ajaysub110/rl-pytorch","url":"https://github.com/ajaysub110/rl-pytorch"},{"title":"watchernyu/spinningup-drl-prototyping","url":"https://github.com/watchernyu/spinningup-drl-prototyping"},{"title":"tliu1997/rnac","url":"https://github.com/tliu1997/rnac"},{"title":"thomashirtz/pytorch-soft-actor-critic","url":"https://github.com/thomashirtz/pytorch-soft-actor-critic"},{"title":"thomashirtz/soft-actor-critic","url":"https://github.com/thomashirtz/soft-actor-critic"},{"title":"ccolas/rl_stats","url":"https://github.com/ccolas/rl_stats"},{"title":"nagisazj/idaq_public","url":"https://github.com/nagisazj/idaq_public"},{"title":"Ipsedo/EvoMotion","url":"https://github.com/Ipsedo/EvoMotion"},{"title":"lucadellalib/sac-beta","url":"https://github.com/lucadellalib/sac-beta"},{"title":"h-aboutalebi/SparceReward","url":"https://github.com/h-aboutalebi/SparceReward"},{"title":"garyzyr001/rethinking-airl","url":"https://github.com/garyzyr001/rethinking-airl"},{"title":"flowersteam/rl_stats","url":"https://github.com/flowersteam/rl_stats"},{"title":"gijskoning/ReproducingCURL","url":"https://github.com/gijskoning/ReproducingCURL"},{"title":"yimingpeng/sac-master","url":"https://github.com/yimingpeng/sac-master"},{"title":"MarsEleven/car_racer_RL","url":"https://github.com/MarsEleven/car_racer_RL"},{"title":"yhisaki/average-reward-drl","url":"https://github.com/yhisaki/average-reward-drl"},{"title":"AmmarFayad/Behavioral-Actor-Critic","url":"https://github.com/AmmarFayad/Behavioral-Actor-Critic"},{"title":"autumnwu/aggressive-q-learning-with-ensembles","url":"https://github.com/autumnwu/aggressive-q-learning-with-ensembles"},{"title":"tarod13/SAC","url":"https://github.com/tarod13/SAC"},{"title":"Steinheilig/Imbiss","url":"https://github.com/Steinheilig/Imbiss"},{"title":"tmjeong1103/RL_with_RAY","url":"https://github.com/tmjeong1103/RL_with_RAY"},{"title":"QuentinVacher-rl/SoftActorCritic-in-Cpp-using-LibTorch","url":"https://github.com/QuentinVacher-rl/SoftActorCritic-in-Cpp-using-LibTorch"},{"title":"SaminYeasar/off_policy_ac","url":"https://github.com/SaminYeasar/off_policy_ac"},{"title":"lollcat/Soft-Actor-Critic","url":"https://github.com/lollcat/Soft-Actor-Critic"},{"title":"donamin/llc","url":"https://github.com/donamin/llc"},{"title":"susan-amin/SparseBaseline1","url":"https://github.com/susan-amin/SparseBaseline1"},{"title":"hyunin-lee/ForecasterSAC","url":"https://github.com/hyunin-lee/ForecasterSAC"},{"title":"cindycia/Atari-SAC-Discrete","url":"https://github.com/cindycia/Atari-SAC-Discrete"},{"title":"mxblr/DeepRLHockey","url":"https://github.com/mxblr/DeepRLHockey"},{"title":"rk1998/robot-sac","url":"https://github.com/rk1998/robot-sac"},{"title":"sunfex/weighted-sac","url":"https://github.com/sunfex/weighted-sac"}],"syntology":{"n":148,"n_ran":91,"n_unverified":57,"n_pointer_only":66}}},{"leaderboard":"/sota/openai-gym-on-ant-v2","slug":"openai-gym-on-ant-v2","dataset":"Ant-v2","dataset_url":null,"rows_in_archive":2,"metrics":["Mean Reward","Average Decisions","Action Repetition"],"first_row_in_archive_order":{"model":"TLA","paper_title":"Optimizing Attention and Cognitive Control Costs Using Temporally-Layered Architectures","paper_url":"/paper/temporally-layered-architecture-for-efficient","paper_date":"2023-05-30","arxiv_id":"2305.18701","code_links":[{"title":"dee0512/Temporally-Layered-Architecture","url":"https://github.com/dee0512/Temporally-Layered-Architecture"}],"syntology":null}},{"leaderboard":"/sota/openai-gym-on-cartpole-v1","slug":"openai-gym-on-cartpole-v1","dataset":"CartPole-v1","dataset_url":null,"rows_in_archive":2,"metrics":["Average Return"],"first_row_in_archive_order":{"model":"Orthogonal decision tree","paper_title":"Evolutionary learning of interpretable decision trees","paper_url":"/paper/evolutionary-learning-of-interpretable","paper_date":"2020-12-14","arxiv_id":"2012.07723","code_links":[{"title":"leocus/ge_q_dts","url":"https://gitlab.com/leocus/ge_q_dts"}],"syntology":null}},{"leaderboard":"/sota/openai-gym-on-halfcheetah-v2","slug":"openai-gym-on-halfcheetah-v2","dataset":"HalfCheetah-v2","dataset_url":null,"rows_in_archive":2,"metrics":["Mean Reward","Average Decisions","Action Repetition"],"first_row_in_archive_order":{"model":"TLA","paper_title":"Optimizing Attention and Cognitive Control Costs Using Temporally-Layered Architectures","paper_url":"/paper/temporally-layered-architecture-for-efficient","paper_date":"2023-05-30","arxiv_id":"2305.18701","code_links":[{"title":"dee0512/Temporally-Layered-Architecture","url":"https://github.com/dee0512/Temporally-Layered-Architecture"}],"syntology":null}},{"leaderboard":"/sota/openai-gym-on-hopper-v2","slug":"openai-gym-on-hopper-v2","dataset":"Hopper-v2","dataset_url":null,"rows_in_archive":2,"metrics":["Mean Reward","Average Decisions","Action Repetition"],"first_row_in_archive_order":{"model":"TLA","paper_title":"Optimizing Attention and Cognitive Control Costs Using Temporally-Layered Architectures","paper_url":"/paper/temporally-layered-architecture-for-efficient","paper_date":"2023-05-30","arxiv_id":"2305.18701","code_links":[{"title":"dee0512/Temporally-Layered-Architecture","url":"https://github.com/dee0512/Temporally-Layered-Architecture"}],"syntology":null}},{"leaderboard":"/sota/openai-gym-on-lunarlander-v2","slug":"openai-gym-on-lunarlander-v2","dataset":"LunarLander-v2","dataset_url":null,"rows_in_archive":2,"metrics":["Average Return"],"first_row_in_archive_order":{"model":"Oblique decision tree","paper_title":"Evolutionary learning of interpretable decision trees","paper_url":"/paper/evolutionary-learning-of-interpretable","paper_date":"2020-12-14","arxiv_id":"2012.07723","code_links":[{"title":"leocus/ge_q_dts","url":"https://gitlab.com/leocus/ge_q_dts"}],"syntology":null}},{"leaderboard":"/sota/openai-gym-on-mountain-car","slug":"openai-gym-on-mountain-car","dataset":"Mountain Car","dataset_url":null,"rows_in_archive":2,"metrics":["Average Return"],"first_row_in_archive_order":{"model":"Orthogonal decision tree","paper_title":"Evolutionary learning of interpretable decision trees","paper_url":"/paper/evolutionary-learning-of-interpretable","paper_date":"2020-12-14","arxiv_id":"2012.07723","code_links":[{"title":"leocus/ge_q_dts","url":"https://gitlab.com/leocus/ge_q_dts"}],"syntology":null}},{"leaderboard":"/sota/openai-gym-on-pendulum-v1","slug":"openai-gym-on-pendulum-v1","dataset":"Pendulum-v1","dataset_url":"/dataset/openai-gym","rows_in_archive":2,"metrics":["Mean Reward","Average Decisions","Action Repetition"],"first_row_in_archive_order":{"model":"TLA with Hierarchical Reward Functions","paper_title":"Creating Hierarchical Dispositions of Needs in an Agent","paper_url":"/paper/creating-hierarchical-dispositions-of-needs","paper_date":"2024-11-23","arxiv_id":"2412.00044","code_links":[{"title":"TofaraMoyo/Heirachical-Reward-Functions","url":"https://github.com/TofaraMoyo/Heirachical-Reward-Functions"}],"syntology":null}},{"leaderboard":"/sota/openai-gym-on-walker2d-v2","slug":"openai-gym-on-walker2d-v2","dataset":"Walker2d-v2","dataset_url":null,"rows_in_archive":2,"metrics":["Mean Reward","Average Decisions","Action Repetition"],"first_row_in_archive_order":{"model":"AWR","paper_title":"Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning","paper_url":"/paper/advantage-weighted-regression-simple-and","paper_date":"2019-10-01","arxiv_id":"1910.00177","code_links":[{"title":"google/trax","url":"https://github.com/google/trax"},{"title":"xbpeng/awr","url":"https://github.com/xbpeng/awr"},{"title":"nvlabs/gbrl_sb3","url":"https://github.com/nvlabs/gbrl_sb3"},{"title":"NitinVishalKulkarni/OfflineReinforcementLearning","url":"https://github.com/NitinVishalKulkarni/OfflineReinforcementLearning"},{"title":"fomorians-oss/awr","url":"https://github.com/fomorians-oss/awr"},{"title":"peisuke/awr","url":"https://github.com/peisuke/awr"}],"syntology":null}},{"leaderboard":"/sota/openai-gym-on-humanoid-v2","slug":"openai-gym-on-humanoid-v2","dataset":"Humanoid-v2","dataset_url":null,"rows_in_archive":1,"metrics":["Average Return"],"first_row_in_archive_order":{"model":"AWR","paper_title":"Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning","paper_url":"/paper/advantage-weighted-regression-simple-and","paper_date":"2019-10-01","arxiv_id":"1910.00177","code_links":[{"title":"google/trax","url":"https://github.com/google/trax"},{"title":"xbpeng/awr","url":"https://github.com/xbpeng/awr"},{"title":"nvlabs/gbrl_sb3","url":"https://github.com/nvlabs/gbrl_sb3"},{"title":"NitinVishalKulkarni/OfflineReinforcementLearning","url":"https://github.com/NitinVishalKulkarni/OfflineReinforcementLearning"},{"title":"fomorians-oss/awr","url":"https://github.com/fomorians-oss/awr"},{"title":"peisuke/awr","url":"https://github.com/peisuke/awr"}],"syntology":null}},{"leaderboard":"/sota/openai-gym-on-inverteddoublependulum-v2","slug":"openai-gym-on-inverteddoublependulum-v2","dataset":"InvertedDoublePendulum-v2","dataset_url":null,"rows_in_archive":1,"metrics":["Mean Reward","Average Decisions","Action Repetition"],"first_row_in_archive_order":{"model":"TLA","paper_title":"Optimizing Attention and Cognitive Control Costs Using Temporally-Layered Architectures","paper_url":"/paper/temporally-layered-architecture-for-efficient","paper_date":"2023-05-30","arxiv_id":"2305.18701","code_links":[{"title":"dee0512/Temporally-Layered-Architecture","url":"https://github.com/dee0512/Temporally-Layered-Architecture"}],"syntology":null}},{"leaderboard":"/sota/openai-gym-on-invertedpendulum-v2","slug":"openai-gym-on-invertedpendulum-v2","dataset":"InvertedPendulum-v2","dataset_url":null,"rows_in_archive":1,"metrics":["Mean Reward","Average Decisions","Action Repetition"],"first_row_in_archive_order":{"model":"TLA","paper_title":"Optimizing Attention and Cognitive Control Costs Using Temporally-Layered Architectures","paper_url":"/paper/temporally-layered-architecture-for-efficient","paper_date":"2023-05-30","arxiv_id":"2305.18701","code_links":[{"title":"dee0512/Temporally-Layered-Architecture","url":"https://github.com/dee0512/Temporally-Layered-Architecture"}],"syntology":null}},{"leaderboard":"/sota/openai-gym-on-mountaincarcontinuous-v0","slug":"openai-gym-on-mountaincarcontinuous-v0","dataset":"MountainCarContinuous-v0","dataset_url":null,"rows_in_archive":1,"metrics":["Mean Reward","Average Decisions","Action Repetition"],"first_row_in_archive_order":{"model":"TLA","paper_title":"Optimizing Attention and Cognitive Control Costs Using Temporally-Layered Architectures","paper_url":"/paper/temporally-layered-architecture-for-efficient","paper_date":"2023-05-30","arxiv_id":"2305.18701","code_links":[{"title":"dee0512/Temporally-Layered-Architecture","url":"https://github.com/dee0512/Temporally-Layered-Architecture"}],"syntology":null}}],"datasets":[{"url":"/dataset/openai-gym","name":"OpenAI Gym","full_name":"OpenAI Gym","num_papers_in_archive":1305},{"url":"/dataset/industrial-benchmark","name":"Industrial Benchmark","full_name":"","num_papers_in_archive":13},{"url":"/dataset/mo-gymnasium","name":"MO-Gymnasium","full_name":"","num_papers_in_archive":8}],"subtasks":[{"url":"/task/acrobot","name":"Acrobot"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":179,"tagged_in_all":382,"items":[{"url":"/paper/proximal-policy-optimization-algorithms","title":"Proximal Policy Optimization Algorithms","date":"2017-07-20","arxiv_id":"1707.06347","repositories_listed":188,"syntology":{"n":176,"n_ran":99,"n_unverified":77,"n_pointer_only":94}},{"url":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","arxiv_id":"1509.02971","repositories_listed":161,"syntology":{"n":306,"n_ran":158,"n_unverified":148,"n_pointer_only":163}},{"url":"/paper/soft-actor-critic-off-policy-maximum-entropy","title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","date":"2018-01-04","arxiv_id":"1801.01290","repositories_listed":86,"syntology":{"n":148,"n_ran":91,"n_unverified":57,"n_pointer_only":66}},{"url":"/paper/addressing-function-approximation-error-in","title":"Addressing Function Approximation Error in Actor-Critic Methods","date":"2018-02-26","arxiv_id":"1802.09477","repositories_listed":67,"syntology":{"n":36,"n_ran":9,"n_unverified":27,"n_pointer_only":20}},{"url":"/paper/multi-goal-reinforcement-learning-challenging","title":"Multi-Goal Reinforcement Learning: Challenging Robotics Environments and Request for Research","date":"2018-02-26","arxiv_id":"1802.09464","repositories_listed":28,"syntology":null},{"url":"/paper/decision-transformer-reinforcement-learning","title":"Decision Transformer: Reinforcement Learning via Sequence Modeling","date":"2021-06-02","arxiv_id":"2106.01345","repositories_listed":20,"syntology":{"n":26,"n_ran":17,"n_unverified":9,"n_pointer_only":6}},{"url":"/paper/advantage-weighted-regression-simple-and","title":"Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning","date":"2019-10-01","arxiv_id":"1910.00177","repositories_listed":6,"syntology":null},{"url":"/paper/deep-recurrent-q-learning-for-partially","title":"Deep Recurrent Q-Learning for Partially Observable MDPs","date":"2015-07-23","arxiv_id":"1507.06527","repositories_listed":5,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/comparing-the-efficacy-of-fine-tuning-and","title":"Comparing the Efficacy of Fine-Tuning and Meta-Learning for Few-Shot Policy Imitation","date":"2023-06-23","arxiv_id":"2306.13554","repositories_listed":4,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-playing-25d","title":"Deep Reinforcement Learning for Playing 2.5D Fighting Games","date":"2018-05-05","arxiv_id":"1805.02070","repositories_listed":4,"syntology":null},{"url":"/paper/learning-to-fly-a-gym-environment-with","title":"Learning to Fly -- a Gym Environment with PyBullet Physics for Reinforcement Learning of Multi-agent Quadcopter Control","date":"2021-03-03","arxiv_id":"2103.02142","repositories_listed":3,"syntology":null},{"url":"/paper/implicit-distributional-reinforcement","title":"Implicit Distributional Reinforcement Learning","date":"2020-07-13","arxiv_id":"2007.06159","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/torchbeast-a-pytorch-platform-for-distributed","title":"TorchBeast: A PyTorch Platform for Distributed RL","date":"2019-10-08","arxiv_id":"1910.03552","repositories_listed":3,"syntology":{"n":9,"n_ran":4,"n_unverified":5,"n_pointer_only":3}},{"url":"/paper/maximum-entropy-regularized-multi-goal","title":"Maximum Entropy-Regularized Multi-Goal Reinforcement Learning","date":"2019-05-21","arxiv_id":"1905.08786","repositories_listed":3,"syntology":null},{"url":"/paper/hierarchical-decision-mamba","title":"Decision Mamba Architectures","date":"2024-05-13","arxiv_id":"2405.07943","repositories_listed":2,"syntology":null},{"url":"/paper/for-sale-state-action-representation-learning-1","title":"For SALE: State-Action Representation Learning for Deep Reinforcement Learning","date":"2023-06-04","arxiv_id":"2306.02451","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/mo-gym-a-library-of-multi-objective","title":"MO-Gym: A Library of Multi-Objective Reinforcement Learning Environments","date":"2022-11-30","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/pyrddlgym-from-rddl-to-gym-environments","title":"pyRDDLGym: From RDDL to Gym Environments","date":"2022-11-11","arxiv_id":"2211.05939","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/cool-mc-a-comprehensive-tool-for","title":"COOL-MC: A Comprehensive Tool for Reinforcement Learning and Model Checking","date":"2022-09-15","arxiv_id":"2209.07133","repositories_listed":2,"syntology":null},{"url":"/paper/bsac-bayesian-strategy-network-based-soft","title":"Bayesian Soft Actor-Critic: A Directed Acyclic Strategy Graph Based Deep Reinforcement Learning","date":"2022-08-11","arxiv_id":"2208.06033","repositories_listed":2,"syntology":null},{"url":"/paper/quantum-deep-reinforcement-learning-for-robot","title":"Quantum Deep Reinforcement Learning for Robot Navigation Tasks","date":"2022-02-24","arxiv_id":"2202.12180","repositories_listed":2,"syntology":{"n":9,"n_ran":0,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/marsexplorer-exploration-of-unknown-terrains","title":"MarsExplorer: Exploration of Unknown Terrains via Deep Reinforcement Learning and Procedurally Generated Environments","date":"2021-07-21","arxiv_id":"2107.09996","repositories_listed":2,"syntology":null},{"url":"/paper/an-open-source-multi-goal-reinforcement","title":"An Open-Source Multi-Goal Reinforcement Learning Environment for Robotic Manipulation with Pybullet","date":"2021-05-12","arxiv_id":"2105.05985","repositories_listed":2,"syntology":null},{"url":"/paper/cropgym-a-reinforcement-learning-environment","title":"CropGym: a Reinforcement Learning Environment for Crop Management","date":"2021-04-09","arxiv_id":"2104.04326","repositories_listed":2,"syntology":null},{"url":"/paper/improving-model-based-reinforcement-learning","title":"Improving Model-Based Reinforcement Learning with Internal State Representations through Self-Supervision","date":"2021-02-10","arxiv_id":"2102.05599","repositories_listed":2,"syntology":null},{"url":"/paper/neurogenetic-programming-framework-for","title":"Neurogenetic Programming Framework for Explainable Reinforcement Learning","date":"2021-02-08","arxiv_id":"2102.04231","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-for-control-of-valves","title":"Reinforcement Learning for Control of Valves","date":"2020-12-29","arxiv_id":"2012.14668","repositories_listed":2,"syntology":null},{"url":"/paper/softgym-benchmarking-deep-reinforcement","title":"SoftGym: Benchmarking Deep Reinforcement Learning for Deformable Object Manipulation","date":"2020-11-14","arxiv_id":"2011.07215","repositories_listed":2,"syntology":null},{"url":"/paper/ecole-a-gym-like-library-for-machine-learning","title":"Ecole: A Gym-like Library for Machine Learning in Combinatorial Optimization Solvers","date":"2020-11-11","arxiv_id":"2011.06069","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/epidemioptim-a-toolbox-for-the-optimization-1","title":"EpidemiOptim: A Toolbox for the Optimization of Control Policies in Epidemiological Models","date":"2020-10-09","arxiv_id":"2010.04452","repositories_listed":2,"syntology":null}],"syntology_records":12,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":1,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}