{"url":"/task/mujoco-games","name":"MuJoCo Games","slug":"mujoco-games","description_markdown":null,"categories":[{"name":"Robots","url":"/area/robots"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":8,"papers_with_code":6,"benchmarks":17,"benchmark_tables_in_archive":17,"benchmark_tables_shown":17,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/mujoco-games-on-ant","slug":"mujoco-games-on-ant","dataset":"Ant","dataset_url":"/dataset/omniverse-isaac-gym","rows_in_archive":3,"metrics":["Average Return"],"first_row_in_archive_order":{"model":"IQ-Learn","paper_title":"IQ-Learn: Inverse soft-Q Learning for Imitation","paper_url":"/paper/iq-learn-inverse-soft-q-learning-for","paper_date":"2021-06-23","arxiv_id":"2106.12142","code_links":[{"title":"Div99/IQ-Learn","url":"https://github.com/Div99/IQ-Learn"},{"title":"robfiras/ls-iq","url":"https://github.com/robfiras/ls-iq"},{"title":"google-deepmind/csil","url":"https://github.com/google-deepmind/csil"},{"title":"edmundmills/basalt-competition","url":"https://github.com/edmundmills/basalt-competition"},{"title":"MilkSilk/masters_thesis","url":"https://github.com/MilkSilk/masters_thesis"}],"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}}},{"leaderboard":"/sota/mujoco-games-on-walker2d","slug":"mujoco-games-on-walker2d","dataset":"Walker2d","dataset_url":null,"rows_in_archive":2,"metrics":["Mean"],"first_row_in_archive_order":{"model":"IQ-Learn","paper_title":"IQ-Learn: Inverse soft-Q Learning for Imitation","paper_url":"/paper/iq-learn-inverse-soft-q-learning-for","paper_date":"2021-06-23","arxiv_id":"2106.12142","code_links":[{"title":"Div99/IQ-Learn","url":"https://github.com/Div99/IQ-Learn"},{"title":"robfiras/ls-iq","url":"https://github.com/robfiras/ls-iq"},{"title":"google-deepmind/csil","url":"https://github.com/google-deepmind/csil"},{"title":"edmundmills/basalt-competition","url":"https://github.com/edmundmills/basalt-competition"},{"title":"MilkSilk/masters_thesis","url":"https://github.com/MilkSilk/masters_thesis"}],"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}}},{"leaderboard":"/sota/mujoco-games-on-ant-v3","slug":"mujoco-games-on-ant-v3","dataset":"Ant-v3","dataset_url":null,"rows_in_archive":1,"metrics":["Average Reward"],"first_row_in_archive_order":{"model":"ParPI","paper_title":"Particle Based Stochastic Policy Optimization","paper_url":"/paper/particle-based-stochastic-policy-optimization","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/mujoco-games-on-halfcheetah","slug":"mujoco-games-on-halfcheetah","dataset":"HalfCheetah","dataset_url":null,"rows_in_archive":1,"metrics":["Mean"],"first_row_in_archive_order":{"model":"POP3D","paper_title":"Policy Optimization With Penalized Point Probability Distance: An Alternative To Proximal Policy Optimization","paper_url":"/paper/policy-optimization-with-penalized-point","paper_date":"2018-07-02","arxiv_id":"1807.00442","code_links":[{"title":"cxxgtxy/POP3D","url":"https://github.com/cxxgtxy/POP3D"},{"title":"paperwithcode/pop3d","url":"https://github.com/paperwithcode/pop3d"}],"syntology":null}},{"leaderboard":"/sota/mujoco-games-on-halfcheetah-v3","slug":"mujoco-games-on-halfcheetah-v3","dataset":"HalfCHeetah-v3","dataset_url":null,"rows_in_archive":1,"metrics":["Average Reward"],"first_row_in_archive_order":{"model":"ParPI","paper_title":"Particle Based Stochastic Policy Optimization","paper_url":"/paper/particle-based-stochastic-policy-optimization","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/mujoco-games-on-hopper","slug":"mujoco-games-on-hopper","dataset":"Hopper","dataset_url":null,"rows_in_archive":1,"metrics":["Mean"],"first_row_in_archive_order":{"model":"POP3D","paper_title":"Policy Optimization With Penalized Point Probability Distance: An Alternative To Proximal Policy Optimization","paper_url":"/paper/policy-optimization-with-penalized-point","paper_date":"2018-07-02","arxiv_id":"1807.00442","code_links":[{"title":"cxxgtxy/POP3D","url":"https://github.com/cxxgtxy/POP3D"},{"title":"paperwithcode/pop3d","url":"https://github.com/paperwithcode/pop3d"}],"syntology":null}},{"leaderboard":"/sota/mujoco-games-on-hopper-v3","slug":"mujoco-games-on-hopper-v3","dataset":"Hopper-v3","dataset_url":null,"rows_in_archive":1,"metrics":["Average Reward"],"first_row_in_archive_order":{"model":"ParPI","paper_title":"Particle Based Stochastic Policy Optimization","paper_url":"/paper/particle-based-stochastic-policy-optimization","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/mujoco-games-on-humanoid-v2","slug":"mujoco-games-on-humanoid-v2","dataset":"Humanoid-v2","dataset_url":null,"rows_in_archive":1,"metrics":["Return"],"first_row_in_archive_order":{"model":"IQ-Learn","paper_title":"IQ-Learn: Inverse soft-Q Learning for Imitation","paper_url":"/paper/iq-learn-inverse-soft-q-learning-for","paper_date":"2021-06-23","arxiv_id":"2106.12142","code_links":[{"title":"Div99/IQ-Learn","url":"https://github.com/Div99/IQ-Learn"},{"title":"robfiras/ls-iq","url":"https://github.com/robfiras/ls-iq"},{"title":"google-deepmind/csil","url":"https://github.com/google-deepmind/csil"},{"title":"edmundmills/basalt-competition","url":"https://github.com/edmundmills/basalt-competition"},{"title":"MilkSilk/masters_thesis","url":"https://github.com/MilkSilk/masters_thesis"}],"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}}},{"leaderboard":"/sota/mujoco-games-on-humanoid-v3","slug":"mujoco-games-on-humanoid-v3","dataset":"Humanoid-v3","dataset_url":null,"rows_in_archive":1,"metrics":["Average Reward"],"first_row_in_archive_order":{"model":"ParPI","paper_title":"Particle Based Stochastic Policy Optimization","paper_url":"/paper/particle-based-stochastic-policy-optimization","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/mujoco-games-on-inverteddoublependulum","slug":"mujoco-games-on-inverteddoublependulum","dataset":"InvertedDoublePendulum","dataset_url":null,"rows_in_archive":1,"metrics":["Mean"],"first_row_in_archive_order":{"model":"POP3D","paper_title":"Policy Optimization With Penalized Point Probability Distance: An Alternative To Proximal Policy Optimization","paper_url":"/paper/policy-optimization-with-penalized-point","paper_date":"2018-07-02","arxiv_id":"1807.00442","code_links":[{"title":"cxxgtxy/POP3D","url":"https://github.com/cxxgtxy/POP3D"},{"title":"paperwithcode/pop3d","url":"https://github.com/paperwithcode/pop3d"}],"syntology":null}},{"leaderboard":"/sota/mujoco-games-on-invertedpendulum","slug":"mujoco-games-on-invertedpendulum","dataset":"InvertedPendulum","dataset_url":null,"rows_in_archive":1,"metrics":["Mean"],"first_row_in_archive_order":{"model":"POP3D","paper_title":"Policy Optimization With Penalized Point Probability Distance: An Alternative To Proximal Policy Optimization","paper_url":"/paper/policy-optimization-with-penalized-point","paper_date":"2018-07-02","arxiv_id":"1807.00442","code_links":[{"title":"cxxgtxy/POP3D","url":"https://github.com/cxxgtxy/POP3D"},{"title":"paperwithcode/pop3d","url":"https://github.com/paperwithcode/pop3d"}],"syntology":null}},{"leaderboard":"/sota/mujoco-games-on-point-maze","slug":"mujoco-games-on-point-maze","dataset":"Point Maze","dataset_url":null,"rows_in_archive":1,"metrics":["Average Return"],"first_row_in_archive_order":{"model":"PEMIRL","paper_title":"Meta-Inverse Reinforcement Learning with Probabilistic Context Variables","paper_url":"/paper/meta-inverse-reinforcement-learning-with","paper_date":"2019-09-20","arxiv_id":"1909.09314","code_links":[{"title":"ermongroup/MetaIRL","url":"https://github.com/ermongroup/MetaIRL"}],"syntology":null}},{"leaderboard":"/sota/mujoco-games-on-reacher","slug":"mujoco-games-on-reacher","dataset":"Reacher","dataset_url":null,"rows_in_archive":1,"metrics":["Mean"],"first_row_in_archive_order":{"model":"POP3D","paper_title":"Policy Optimization With Penalized Point Probability Distance: An Alternative To Proximal Policy Optimization","paper_url":"/paper/policy-optimization-with-penalized-point","paper_date":"2018-07-02","arxiv_id":"1807.00442","code_links":[{"title":"cxxgtxy/POP3D","url":"https://github.com/cxxgtxy/POP3D"},{"title":"paperwithcode/pop3d","url":"https://github.com/paperwithcode/pop3d"}],"syntology":null}},{"leaderboard":"/sota/mujoco-games-on-sawyer-pusher","slug":"mujoco-games-on-sawyer-pusher","dataset":"Sawyer Pusher","dataset_url":null,"rows_in_archive":1,"metrics":["Average Return"],"first_row_in_archive_order":{"model":"PEMIRL","paper_title":"Meta-Inverse Reinforcement Learning with Probabilistic Context Variables","paper_url":"/paper/meta-inverse-reinforcement-learning-with","paper_date":"2019-09-20","arxiv_id":"1909.09314","code_links":[{"title":"ermongroup/MetaIRL","url":"https://github.com/ermongroup/MetaIRL"}],"syntology":null}},{"leaderboard":"/sota/mujoco-games-on-sweeper","slug":"mujoco-games-on-sweeper","dataset":"Sweeper","dataset_url":null,"rows_in_archive":1,"metrics":["Average Return"],"first_row_in_archive_order":{"model":"PEMIRL","paper_title":"Meta-Inverse Reinforcement Learning with Probabilistic Context Variables","paper_url":"/paper/meta-inverse-reinforcement-learning-with","paper_date":"2019-09-20","arxiv_id":"1909.09314","code_links":[{"title":"ermongroup/MetaIRL","url":"https://github.com/ermongroup/MetaIRL"}],"syntology":null}},{"leaderboard":"/sota/mujoco-games-on-swimmer","slug":"mujoco-games-on-swimmer","dataset":"Swimmer","dataset_url":null,"rows_in_archive":1,"metrics":["Mean"],"first_row_in_archive_order":{"model":"POP3D","paper_title":"Policy Optimization With Penalized Point Probability Distance: An Alternative To Proximal Policy Optimization","paper_url":"/paper/policy-optimization-with-penalized-point","paper_date":"2018-07-02","arxiv_id":"1807.00442","code_links":[{"title":"cxxgtxy/POP3D","url":"https://github.com/cxxgtxy/POP3D"},{"title":"paperwithcode/pop3d","url":"https://github.com/paperwithcode/pop3d"}],"syntology":null}},{"leaderboard":"/sota/mujoco-games-on-walker2d-v3","slug":"mujoco-games-on-walker2d-v3","dataset":"Walker2d-v3","dataset_url":null,"rows_in_archive":1,"metrics":["Average Reward"],"first_row_in_archive_order":{"model":"ParPI","paper_title":"Particle Based Stochastic Policy Optimization","paper_url":"/paper/particle-based-stochastic-policy-optimization","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/omniverse-isaac-gym","name":"Omniverse Isaac Gym","full_name":"","num_papers_in_archive":240},{"url":"/dataset/mo-gymnasium","name":"MO-Gymnasium","full_name":"","num_papers_in_archive":8},{"url":"/dataset/il-datasets","name":"IL-Datasets","full_name":"Imitation Datasets","num_papers_in_archive":2}],"subtasks":[{"url":"/task/d4rl","name":"D4RL"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":6,"of":6,"tagged_in_all":8,"items":[{"url":"/paper/iq-learn-inverse-soft-q-learning-for","title":"IQ-Learn: Inverse soft-Q Learning for Imitation","date":"2021-06-23","arxiv_id":"2106.12142","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/a-unified-approach-to-reinforcement-learning","title":"A Unified Approach to Reinforcement Learning, Quantal Response Equilibria, and Two-Player Zero-Sum Games","date":"2022-06-12","arxiv_id":"2206.05825","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/rl-unplugged-benchmarks-for-offline","title":"RL Unplugged: A Suite of Benchmarks for Offline Reinforcement Learning","date":"2020-06-24","arxiv_id":"2006.13888","repositories_listed":2,"syntology":null},{"url":"/paper/ls-iq-implicit-reward-regularization-for","title":"LS-IQ: Implicit Reward Regularization for Inverse Reinforcement Learning","date":"2023-03-01","arxiv_id":"2303.00599","repositories_listed":1,"syntology":null},{"url":"/paper/edge-explaining-deep-reinforcement-learning","title":"EDGE: Explaining Deep Reinforcement Learning Policies","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/human-preference-scaling-with-demonstrations","title":"Weak Human Preference Supervision For Deep Reinforcement Learning","date":"2020-07-25","arxiv_id":"2007.12904","repositories_listed":1,"syntology":null}],"syntology_records":2,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}