{"url":"/task/unsupervised-reinforcement-learning","name":"Unsupervised Reinforcement Learning","slug":"unsupervised-reinforcement-learning","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":57,"papers_with_code":29,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":2,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/urlb","name":"URLB","full_name":"Unsupervised Reinforcement Learning Benchmark","num_papers_in_archive":30},{"url":"/dataset/bipedal-skills","name":"bipedal-skills","full_name":"Bipedal Skills Benchmark for Reinforcement Learning","num_papers_in_archive":2}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":29,"of":29,"tagged_in_all":57,"items":[{"url":"/paper/exploration-by-random-network-distillation","title":"Exploration by Random Network Distillation","date":"2018-10-30","arxiv_id":"1810.12894","repositories_listed":22,"syntology":{"n":43,"n_ran":26,"n_unverified":17,"n_pointer_only":15}},{"url":"/paper/curiosity-driven-exploration-by-self","title":"Curiosity-driven Exploration by Self-supervised Prediction","date":"2017-05-15","arxiv_id":"1705.05363","repositories_listed":13,"syntology":{"n":32,"n_ran":16,"n_unverified":16,"n_pointer_only":14}},{"url":"/paper/mastering-visual-continuous-control-improved","title":"Mastering Visual Continuous Control: Improved Data-Augmented Reinforcement Learning","date":"2021-07-20","arxiv_id":"2107.09645","repositories_listed":8,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/diversity-is-all-you-need-learning-skills","title":"Diversity is All You Need: Learning Skills without a Reward Function","date":"2018-02-16","arxiv_id":"1802.06070","repositories_listed":4,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/zero-shot-whole-body-humanoid-control-via","title":"Zero-Shot Whole-Body Humanoid Control via Behavioral Foundation Models","date":"2025-04-15","arxiv_id":"2504.11054","repositories_listed":2,"syntology":{"n":31,"n_ran":16,"n_unverified":15,"n_pointer_only":31}},{"url":"/paper/unsupervised-reinforcement-learning-in","title":"Unsupervised Reinforcement Learning in Multiple Environments","date":"2021-12-16","arxiv_id":"2112.08746","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/emergent-real-world-robotic-skills-via","title":"Emergent Real-World Robotic Skills via Unsupervised Off-Policy Reinforcement Learning","date":"2020-04-27","arxiv_id":"2004.12974","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/self-supervised-exploration-via-disagreement","title":"Self-Supervised Exploration via Disagreement","date":"2019-06-10","arxiv_id":"1906.04161","repositories_listed":2,"syntology":null},{"url":"/paper/surprise-adaptive-intrinsic-motivation-for","title":"Surprise-Adaptive Intrinsic Motivation for Unsupervised Reinforcement Learning","date":"2024-05-27","arxiv_id":"2405.17243","repositories_listed":1,"syntology":null},{"url":"/paper/m2curl-sample-efficient-multimodal","title":"M2CURL: Sample-Efficient Multimodal Reinforcement Learning via Self-Supervised Representation Learning for Robotic Manipulation","date":"2024-01-30","arxiv_id":"2401.17032","repositories_listed":1,"syntology":null},{"url":"/paper/metra-scalable-unsupervised-rl-with-metric","title":"METRA: Scalable Unsupervised RL with Metric-Aware Abstraction","date":"2023-10-13","arxiv_id":"2310.08887","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/comsd-balancing-behavioral-quality-and","title":"ComSD: Balancing Behavioral Quality and Diversity in Unsupervised Skill Discovery","date":"2023-09-29","arxiv_id":"2309.17203","repositories_listed":1,"syntology":null},{"url":"/paper/crc-rl-a-novel-visual-feature-representation","title":"CRC-RL: A Novel Visual Feature Representation Architecture for Unsupervised Reinforcement Learning","date":"2023-01-31","arxiv_id":"2301.13473","repositories_listed":1,"syntology":null},{"url":"/paper/choreographer-learning-and-adapting-skills-in","title":"Choreographer: Learning and Adapting Skills in Imagination","date":"2022-11-23","arxiv_id":"2211.13350","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/skill-based-reinforcement-learning-with","title":"Skill-Based Reinforcement Learning with Intrinsic Reward Matching","date":"2022-10-14","arxiv_id":"2210.07426","repositories_listed":1,"syntology":null},{"url":"/paper/a-mixture-of-surprises-for-unsupervised","title":"A Mixture of Surprises for Unsupervised Reinforcement Learning","date":"2022-10-13","arxiv_id":"2210.06702","repositories_listed":1,"syntology":{"n":12,"n_ran":1,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/unsupervised-model-based-pre-training-for","title":"Mastering the Unsupervised Reinforcement Learning Benchmark from Pixels","date":"2022-09-24","arxiv_id":"2209.12016","repositories_listed":1,"syntology":{"n":24,"n_ran":17,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/impact-makes-a-sound-and-sound-makes-an","title":"Impact Makes a Sound and Sound Makes an Impact: Sound Guides Representations and Explorations","date":"2022-08-04","arxiv_id":"2208.02680","repositories_listed":1,"syntology":null},{"url":"/paper/ase-large-scale-reusable-adversarial-skill","title":"ASE: Large-Scale Reusable Adversarial Skill Embeddings for Physically Simulated Characters","date":"2022-05-04","arxiv_id":"2205.01906","repositories_listed":1,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/cic-contrastive-intrinsic-control-for-1","title":"CIC: Contrastive Intrinsic Control for Unsupervised Skill Discovery","date":"2022-02-01","arxiv_id":"2202.00161","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/urlb-unsupervised-reinforcement-learning","title":"URLB: Unsupervised Reinforcement Learning Benchmark","date":"2021-10-28","arxiv_id":"2110.15191","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/the-information-geometry-of-unsupervised","title":"The Information Geometry of Unsupervised Reinforcement Learning","date":"2021-10-06","arxiv_id":"2110.02719","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-multi-latent-space-reinforcement","title":"Unsupervised multi-latent space reinforcement learning framework for video summarization in ultrasound imaging","date":"2021-09-03","arxiv_id":"2109.01309","repositories_listed":1,"syntology":null},{"url":"/paper/explore-and-control-with-adversarial-surprise","title":"Explore and Control with Adversarial Surprise","date":"2021-07-12","arxiv_id":"2107.07394","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/behavior-from-the-void-unsupervised-active","title":"Behavior From the Void: Unsupervised Active Pre-Training","date":"2021-03-08","arxiv_id":"2103.04551","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/reinforcement-learning-with-prototypical","title":"Reinforcement Learning with Prototypical Representations","date":"2021-02-22","arxiv_id":"2102.11271","repositories_listed":1,"syntology":null},{"url":"/paper/smirl-surprise-minimizing-rl-in-dynamic","title":"SMiRL: Surprise Minimizing Reinforcement Learning in Unstable Environments","date":"2019-12-11","arxiv_id":"1912.05510","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-exploration-via-state-marginal","title":"Efficient Exploration via State Marginal Matching","date":"2019-06-12","arxiv_id":"1906.05274","repositories_listed":1,"syntology":null},{"url":"/paper/variational-intrinsic-control","title":"Variational Intrinsic Control","date":"2016-11-22","arxiv_id":"1611.07507","repositories_listed":1,"syntology":null}],"syntology_records":16,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}