{"url":"/task/meta-reinforcement-learning","name":"Meta Reinforcement Learning","slug":"meta-reinforcement-learning","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":278,"papers_with_code":103,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":2,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/meta-world-benchmark","name":"Meta-World Benchmark","full_name":"","num_papers_in_archive":73},{"url":"/dataset/mikasa-robo-dataset","name":"MIKASA-Robo Dataset","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":103,"tagged_in_all":278,"items":[{"url":"/paper/meta-world-a-benchmark-and-evaluation-for","title":"Meta-World: A Benchmark and Evaluation for Multi-Task and Meta Reinforcement Learning","date":"2019-10-24","arxiv_id":"1910.10897","repositories_listed":9,"syntology":{"n":9,"n_ran":0,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/learning-to-reinforcement-learn","title":"Learning to reinforcement learn","date":"2016-11-17","arxiv_id":"1611.05763","repositories_listed":9,"syntology":{"n":15,"n_ran":3,"n_unverified":12,"n_pointer_only":12}},{"url":"/paper/efficient-off-policy-meta-reinforcement","title":"Efficient Off-Policy Meta-Reinforcement Learning via Probabilistic Context Variables","date":"2019-03-19","arxiv_id":"1903.08254","repositories_listed":7,"syntology":{"n":9,"n_ran":6,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/some-considerations-on-learning-to-explore","title":"Some Considerations on Learning to Explore via Meta-Reinforcement Learning","date":"2018-03-03","arxiv_id":"1803.01118","repositories_listed":7,"syntology":null},{"url":"/paper/promp-proximal-meta-policy-search","title":"ProMP: Proximal Meta-Policy Search","date":"2018-10-16","arxiv_id":"1810.06784","repositories_listed":6,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/pixelsnail-an-improved-autoregressive","title":"PixelSNAIL: An Improved Autoregressive Generative Model","date":"2017-12-28","arxiv_id":"1712.09763","repositories_listed":6,"syntology":{"n":7,"n_ran":6,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/diversity-is-all-you-need-learning-skills","title":"Diversity is All You Need: Learning Skills without a Reward Function","date":"2018-02-16","arxiv_id":"1802.06070","repositories_listed":4,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/evolving-reservoirs-for-meta-reinforcement","title":"Evolving Reservoirs for Meta Reinforcement Learning","date":"2023-12-09","arxiv_id":"2312.06695","repositories_listed":3,"syntology":null},{"url":"/paper/2505-11289","title":"Meta-World+: An Improved, Standardized, RL Benchmark","date":"2025-05-16","arxiv_id":"2505.11289","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/xland-minigrid-scalable-meta-reinforcement","title":"XLand-MiniGrid: Scalable Meta-Reinforcement Learning Environments in JAX","date":"2023-12-19","arxiv_id":"2312.12044","repositories_listed":2,"syntology":null},{"url":"/paper/context-meta-reinforcement-learning-via","title":"Context Meta-Reinforcement Learning via Neuromodulation","date":"2021-10-30","arxiv_id":"2111.00134","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/learn2learn-a-library-for-meta-learning","title":"learn2learn: A Library for Meta-Learning Research","date":"2020-08-27","arxiv_id":"2008.12284","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/offline-meta-reinforcement-learning-with","title":"Offline Meta-Reinforcement Learning with Advantage Weighting","date":"2020-08-13","arxiv_id":"2008.06043","repositories_listed":2,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":5}},{"url":"/paper/explore-then-execute-adapting-without-rewards","title":"Decoupling Exploration and Exploitation for Meta-Reinforcement Learning without Sacrifices","date":"2020-08-06","arxiv_id":"2008.02790","repositories_listed":2,"syntology":null},{"url":"/paper/multi-task-reinforcement-learning-as-a-hidden","title":"Learning Robust State Abstractions for Hidden-Parameter Block MDPs","date":"2020-07-14","arxiv_id":"2007.07206","repositories_listed":2,"syntology":null},{"url":"/paper/model-based-meta-reinforcement-learning-for","title":"Model-Based Meta-Reinforcement Learning for Flight with Suspended Payloads","date":"2020-04-23","arxiv_id":"2004.11345","repositories_listed":2,"syntology":null},{"url":"/paper/meta-q-learning","title":"Meta-Q-Learning","date":"2019-09-30","arxiv_id":"1910.00125","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/meta-reinforcement-learning-with-task","title":"Meta Reinforcement Learning with Task Embedding and Shared Policy","date":"2019-05-16","arxiv_id":"1905.06527","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-learn-how-to-learn-self-adaptive","title":"Learning to Learn How to Learn: Self-Adaptive Visual Navigation Using Meta-Learning","date":"2018-12-03","arxiv_id":"1812.00971","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/learning-to-adapt-in-dynamic-real-world","title":"Learning to Adapt in Dynamic, Real-World Environments Through Meta-Reinforcement Learning","date":"2018-03-30","arxiv_id":"1803.11347","repositories_listed":2,"syntology":null},{"url":"/paper/meta-reinforcement-learning-of-structured","title":"Meta-Reinforcement Learning of Structured Exploration Strategies","date":"2018-02-20","arxiv_id":"1802.07245","repositories_listed":2,"syntology":null},{"url":"/paper/learning-task-belief-similarity-with-latent","title":"Learning Task Belief Similarity with Latent Dynamics for Meta-Reinforcement Learning","date":"2025-06-24","arxiv_id":"2506.19785","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-meta-reinforcement-learning-with","title":"Bayesian Meta-Reinforcement Learning with Laplace Variational Recurrent Networks","date":"2025-05-24","arxiv_id":"2505.18591","repositories_listed":1,"syntology":null},{"url":"/paper/task-aware-virtual-training-enhancing","title":"Task-Aware Virtual Training: Enhancing Generalization in Meta-Reinforcement Learning for Out-of-Distribution Tasks","date":"2025-02-05","arxiv_id":"2502.02834","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/coreset-based-task-selection-for-sample","title":"Coreset-Based Task Selection for Sample-Efficient Meta-Reinforcement Learning","date":"2025-02-04","arxiv_id":"2502.02332","repositories_listed":1,"syntology":null},{"url":"/paper/entropy-regularized-task-representation","title":"Entropy Regularized Task Representation Learning for Offline Meta-Reinforcement Learning","date":"2024-12-19","arxiv_id":"2412.14834","repositories_listed":1,"syntology":null},{"url":"/paper/towards-generalizable-autonomous-penetration","title":"Mind the Gap: Towards Generalizable Autonomous Penetration Testing via Domain Randomization and Meta-Reinforcement Learning","date":"2024-12-05","arxiv_id":"2412.04078","repositories_listed":1,"syntology":null},{"url":"/paper/amago-2-breaking-the-multi-task-barrier-in","title":"AMAGO-2: Breaking the Multi-Task Barrier in Meta-Reinforcement Learning with Transformers","date":"2024-11-17","arxiv_id":"2411.11188","repositories_listed":1,"syntology":{"n":10,"n_ran":2,"n_unverified":8,"n_pointer_only":1}},{"url":"/paper/enabling-adaptive-agent-training-in-open","title":"Enabling Adaptive Agent Training in Open-Ended Simulators by Targeting Diversity","date":"2024-11-07","arxiv_id":"2411.04466","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/falcon-feedback-driven-adaptive-long-short","title":"FALCON: Feedback-driven Adaptive Long/short-term memory reinforced Coding Optimization system","date":"2024-10-28","arxiv_id":"2410.21349","repositories_listed":1,"syntology":null}],"syntology_records":15,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}