{"url":"/task/mujoco","name":"MuJoCo","slug":"mujoco","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":677,"papers_with_code":293,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":2,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/mujoco","name":"MuJoCo","full_name":"","num_papers_in_archive":1638},{"url":"/dataset/neorl","name":"NeoRL","full_name":"","num_papers_in_archive":10}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":293,"tagged_in_all":677,"items":[{"url":"/paper/simple-random-search-provides-a-competitive","title":"Simple random search provides a competitive approach to reinforcement learning","date":"2018-03-19","arxiv_id":"1803.07055","repositories_listed":26,"syntology":{"n":15,"n_ran":3,"n_unverified":12,"n_pointer_only":3}},{"url":"/paper/the-starcraft-multi-agent-challenge","title":"The StarCraft Multi-Agent Challenge","date":"2019-02-11","arxiv_id":"1902.04043","repositories_listed":23,"syntology":{"n":15,"n_ran":6,"n_unverified":9,"n_pointer_only":13}},{"url":"/paper/evolution-strategies-as-a-scalable","title":"Evolution Strategies as a Scalable Alternative to Reinforcement Learning","date":"2017-03-10","arxiv_id":"1703.03864","repositories_listed":23,"syntology":{"n":29,"n_ran":7,"n_unverified":22,"n_pointer_only":1}},{"url":"/paper/trust-region-policy-optimisation-in-multi","title":"Trust Region Policy Optimisation in Multi-Agent Reinforcement Learning","date":"2021-09-23","arxiv_id":"2109.11251","repositories_listed":11,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/neural-network-dynamics-for-model-based-deep","title":"Neural Network Dynamics for Model-Based Deep Reinforcement Learning with Model-Free Fine-Tuning","date":"2017-08-08","arxiv_id":"1708.02596","repositories_listed":9,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/deepmind-control-suite","title":"DeepMind Control Suite","date":"2018-01-02","arxiv_id":"1801.00690","repositories_listed":8,"syntology":null},{"url":"/paper/scalable-trust-region-method-for-deep","title":"Scalable trust-region method for deep reinforcement learning using Kronecker-factored approximation","date":"2017-08-17","arxiv_id":"1708.05144","repositories_listed":8,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/robosuite-a-modular-simulation-framework-and","title":"robosuite: A Modular Simulation Framework and Benchmark for Robot Learning","date":"2020-09-25","arxiv_id":"2009.12293","repositories_listed":7,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":1}},{"url":"/paper/randomized-ensembled-double-q-learning-1","title":"Randomized Ensembled Double Q-Learning: Learning Fast Without a Model","date":"2021-01-15","arxiv_id":"2101.05982","repositories_listed":6,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/sqil-imitation-learning-via-regularized","title":"SQIL: Imitation Learning via Reinforcement Learning with Sparse Rewards","date":"2019-05-27","arxiv_id":"1905.11108","repositories_listed":5,"syntology":null},{"url":"/paper/comparing-the-efficacy-of-fine-tuning-and","title":"Comparing the Efficacy of Fine-Tuning and Meta-Learning for Few-Shot Policy Imitation","date":"2023-06-23","arxiv_id":"2306.13554","repositories_listed":4,"syntology":null},{"url":"/paper/multi-agent-constrained-policy-optimisation","title":"Multi-Agent Constrained Policy Optimisation","date":"2021-10-06","arxiv_id":"2110.02793","repositories_listed":4,"syntology":{"n":7,"n_ran":1,"n_unverified":6,"n_pointer_only":1}},{"url":"/paper/improving-sample-efficiency-in-model-free-1","title":"Improving Sample Efficiency in Model-Free Reinforcement Learning from Images","date":"2019-10-02","arxiv_id":"1910.01741","repositories_listed":4,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/self-imitation-learning","title":"Self-Imitation Learning","date":"2018-06-14","arxiv_id":"1806.05635","repositories_listed":4,"syntology":{"n":6,"n_ran":1,"n_unverified":5,"n_pointer_only":3}},{"url":"/paper/balance-reward-and-safety-optimization-for","title":"Balance Reward and Safety Optimization for Safe Reinforcement Learning: A Perspective of Gradient Manipulation","date":"2024-05-02","arxiv_id":"2405.01677","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/envpool-a-highly-parallel-reinforcement","title":"EnvPool: A Highly Parallel Reinforcement Learning Environment Execution Engine","date":"2022-06-21","arxiv_id":"2206.10558","repositories_listed":3,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/rank-the-episodes-a-simple-approach-for-1","title":"Rank the Episodes: A Simple Approach for Exploration in Procedurally-Generated Environments","date":"2021-01-20","arxiv_id":"2101.08152","repositories_listed":3,"syntology":{"n":28,"n_ran":14,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/reinforcement-learning-with-random-delays-1","title":"Reinforcement Learning with Random Delays","date":"2020-10-06","arxiv_id":"2010.02966","repositories_listed":3,"syntology":{"n":9,"n_ran":2,"n_unverified":7,"n_pointer_only":1}},{"url":"/paper/how-to-learn-a-useful-critic-model-based","title":"How to Learn a Useful Critic? Model-based Action-Gradient-Estimator Policy Optimization","date":"2020-04-29","arxiv_id":"2004.14309","repositories_listed":3,"syntology":{"n":6,"n_ran":1,"n_unverified":5,"n_pointer_only":1}},{"url":"/paper/deep-multi-agent-reinforcement-learning-for","title":"FACMAC: Factored Multi-Agent Centralised Policy Gradients","date":"2020-03-14","arxiv_id":"2003.06709","repositories_listed":3,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/addressing-value-estimation-errors-in","title":"Distributional Soft Actor-Critic: Off-Policy Reinforcement Learning for Addressing Value Estimation Errors","date":"2020-01-09","arxiv_id":"2001.02811","repositories_listed":3,"syntology":null},{"url":"/paper/varibad-a-very-good-method-for-bayes-adaptive-1","title":"VariBAD: A Very Good Method for Bayes-Adaptive Deep RL via Meta-Learning","date":"2019-10-18","arxiv_id":"1910.08348","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/boosting-soft-actor-critic-emphasizing-recent","title":"Boosting Soft Actor-Critic: Emphasizing Recent Experience without Forgetting the Past","date":"2019-06-10","arxiv_id":"1906.04009","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/extrapolating-beyond-suboptimal","title":"Extrapolating Beyond Suboptimal Demonstrations via Inverse Reinforcement Learning from Observations","date":"2019-04-12","arxiv_id":"1904.06387","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/learnings-options-end-to-end-for-continuous","title":"Learnings Options End-to-End for Continuous Action Tasks","date":"2017-11-30","arxiv_id":"1712.00004","repositories_listed":3,"syntology":null},{"url":"/paper/a-bayesian-approach-to-robust-inverse","title":"A Bayesian Approach to Robust Inverse Reinforcement Learning","date":"2023-09-15","arxiv_id":"2309.08571","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/explaining-rl-decisions-with-trajectories","title":"Explaining RL Decisions with Trajectories","date":"2023-05-06","arxiv_id":"2305.04073","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/understanding-expertise-through","title":"When Demonstrations Meet Generative World Models: A Maximum Likelihood Framework for Offline Inverse Reinforcement Learning","date":"2023-02-15","arxiv_id":"2302.07457","repositories_listed":2,"syntology":null},{"url":"/paper/locally-constrained-policy-optimization-for","title":"Online Reinforcement Learning in Non-Stationary Context-Driven Environments","date":"2023-02-04","arxiv_id":"2302.02182","repositories_listed":2,"syntology":{"n":6,"n_ran":1,"n_unverified":5,"n_pointer_only":5}},{"url":"/paper/on-the-effect-of-pre-training-for-transformer","title":"On the Effect of Pre-training for Transformer in Different Modality on Offline Reinforcement Learning","date":"2022-11-17","arxiv_id":"2211.09817","repositories_listed":2,"syntology":{"n":14,"n_ran":6,"n_unverified":8,"n_pointer_only":0}}],"syntology_records":24,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}