{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/meta-reinforcement-learning/papers/3","list_of":"/task/meta-reinforcement-learning","task":"Meta Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":3,"rows_per_page":100,"rows":[201,278],"of":278,"counts":{"archive_papers_tagged":278,"with_a_code_link":103,"where_syntology_ran_a_sample":39,"not_listed_spam_title":0,"listed":278,"listed_where_code_ran":39,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":32,"every_run_a_failure_of_syntologys_instrument":7,"listed_with_a_run_with_no_instrument_failure":32,"listed_every_run_a_failure_of_syntologys_instrument":7,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/meta-reinforcement-learning","prev":"/task/meta-reinforcement-learning/papers/2","next":null,"papers":[{"url":null,"slug":"provable-hierarchy-based-meta-reinforcement-1","title":"Provable Hierarchy-Based Meta-Reinforcement Learning","date":"2021-10-18","arxiv_id":"2110.09507","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-are-meta-reinforcement-learners","title":"Transformers are Meta-Reinforcement Learners","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"introducing-symmetries-to-black-box-meta","title":"Introducing Symmetries to Black Box Meta Reinforcement Learning","date":"2021-09-22","arxiv_id":"2109.10781","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrated-and-adaptive-guidance-and-control","title":"Integrated and Adaptive Guidance and Control for Endoatmospheric Missiles via Reinforcement Learning","date":"2021-09-08","arxiv_id":"2109.03880","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-safe-model-based-meta-reinforcement","title":"Provably Safe Model-Based Meta Reinforcement Learning: An Abstraction-Based Approach","date":"2021-09-03","arxiv_id":"2109.01255","repositories_listed":0,"syntology":null},{"url":null,"slug":"prior-is-all-you-need-to-improve-the","title":"Improved Robustness and Safety for Pre-Adaptation of Meta Reinforcement Learning with Prior Regularization","date":"2021-08-19","arxiv_id":"2108.08448","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-mastering","title":"Meta-Reinforcement Learning for Mastering Multiple Skills and Generalizing across Environments in Text-based Games","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptation-of-quadruped-robot-locomotion-with","title":"Adaptation of Quadruped Robot Locomotion with Meta-Learning","date":"2021-07-08","arxiv_id":"2107.03741","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-heuristic","title":"Meta-Reinforcement Learning for Heuristic Planning","date":"2021-07-06","arxiv_id":"2107.02603","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-predictive-representation-for","title":"Disentangled Predictive Representation for Meta-Reinforcement Learning","date":"2021-06-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-fast","title":"Meta Reinforcement Learning for Fast Adaptation of Hierarchical Policies","date":"2021-05-21","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-the-step-size-in-policy","title":"Meta Learning the Step Size in Policy Gradient Methods","date":"2021-05-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"estimating-disentangled-belief-about-hidden","title":"Estimating Disentangled Belief about Hidden State and Hidden Task for Meta-RL","date":"2021-05-14","arxiv_id":"2105.06660","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-meta-reinforcement-learning-to-bridge","title":"Using Meta Reinforcement Learning to Bridge the Gap between Simulation and Experiment in Energy Demand Response","date":"2021-04-29","arxiv_id":"2104.14670","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-is-going-on-inside-recurrent-meta","title":"What is Going on Inside Recurrent Meta Reinforcement Learning Agents?","date":"2021-04-29","arxiv_id":"2104.14644","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-adversarial-training-for-meta","title":"Adaptive Adversarial Training for Meta Reinforcement Learning","date":"2021-04-27","arxiv_id":"2104.13302","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-meta-reinforcement-learning-approach-to","title":"A Meta-Reinforcement Learning Approach to Process Control","date":"2021-03-25","arxiv_id":"2103.14060","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-maml-prioritization-task-buffer-with","title":"Robust MAML: Prioritization task buffer with adaptive learning process for model-agnostic meta-learning","date":"2021-03-15","arxiv_id":"2103.08233","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-for-multi-agent-communication","title":"Meta Learning for Multi-agent Communication","date":"2021-03-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-context-based-meta-reinforcement","title":"Improving Context-Based Meta-Reinforcement Learning with Self-Supervised Trajectory Contrastive Learning","date":"2021-03-10","arxiv_id":"2103.06386","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-meta-reinforcement-learning-using","title":"Model-based Meta Reinforcement Learning using Graph Structured Surrogate Models","date":"2021-02-16","arxiv_id":"2102.08291","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-adaptive","title":"Meta-Reinforcement Learning for Adaptive Motor Control in Changing Robot Dynamics and Environments","date":"2021-01-19","arxiv_id":"2101.07599","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-and-fast-adaptation-for-grid","title":"Learning and Fast Adaptation for Grid Emergency Control via Deep Meta Reinforcement Learning","date":"2021-01-13","arxiv_id":"2101.05317","repositories_listed":0,"syntology":null},{"url":null,"slug":"linear-representation-meta-reinforcement-1","title":"Linear Representation Meta-Reinforcement Learning for Instant Adaptation","date":"2021-01-12","arxiv_id":"2101.04750","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-meta-reinforcement-learning-based","title":"Off-Policy Meta-Reinforcement Learning Based on Feature Embedding Spaces","date":"2021-01-06","arxiv_id":"2101.01883","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-meta-reinforcement-learning","title":"Interpretable Meta-Reinforcement Learning with Actor-Critic Method","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsically-guided-exploration-in-meta","title":"Intrinsically Guided Exploration in Meta Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-with-informed","title":"Meta-Reinforcement Learning With Informed Policy Regularization","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"peril-probabilistic-embeddings-for-hybrid","title":"PERIL: Probabilistic Embeddings for hybrid Meta-Reinforcement and Imitation Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-weighed-policy-sampling-for-meta","title":"Performance-Weighed Policy Sampling for Meta-Reinforcement Learning","date":"2020-12-10","arxiv_id":"2012.06016","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-meta-learning-for-data-efficient","title":"Double Meta-Learning for Data Efficient Policy Optimization in Non-Stationary Environments","date":"2020-11-21","arxiv_id":"2011.10714","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-theoretic-task-selection-for-meta","title":"Information-theoretic Task Selection for Meta-Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.01054","repositories_listed":0,"syntology":null},{"url":null,"slug":"characterizing-policy-divergence-for","title":"Characterizing Policy Divergence for Personalized Meta-Reinforcement Learning","date":"2020-10-09","arxiv_id":"2010.04816","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-based-bayesian-meta-reinforcement","title":"Bayesian Meta-reinforcement Learning for Traffic Signal Control","date":"2020-10-01","arxiv_id":"2010.00163","repositories_listed":0,"syntology":null},{"url":null,"slug":"complementary-meta-reinforcement-learning-for","title":"Complementary Meta-Reinforcement Learning for Fault-Adaptive Control","date":"2020-09-26","arxiv_id":"2009.12634","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalight-improving-environment","title":"GeneraLight: Improving Environment Generalization of Traffic Signal Control via Meta Reinforcement Learning","date":"2020-09-17","arxiv_id":"2009.08052","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-weighted-policy-learning-and","title":"Importance Weighted Policy Learning and Adaptation","date":"2020-09-10","arxiv_id":"2009.04875","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-based-lane-change","title":"Meta Reinforcement Learning-Based Lane Change Strategy for Autonomous Vehicles","date":"2020-08-28","arxiv_id":"2008.12451","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-model-based-meta-reinforcement-learning","title":"Safe Active Dynamics Learning and Control: A Sequential Exploration-Exploitation Framework","date":"2020-08-26","arxiv_id":"2008.11700","repositories_listed":0,"syntology":null},{"url":"/paper/catch-context-based-meta-reinforcement","slug":"catch-context-based-meta-reinforcement","title":"CATCH: Context-based Meta Reinforcement Learning for Transferrable Architecture Search","date":"2020-07-18","arxiv_id":"2007.09380","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-brief-look-at-generalization-in-visual-meta","title":"A Brief Look at Generalization in Visual Meta-Reinforcement Learning","date":"2020-06-12","arxiv_id":"2006.07262","repositories_listed":0,"syntology":null},{"url":null,"slug":"explore-then-execute-adapting-without-rewards-1","title":"Explore then Execute: Adapting without Rewards via Factorized Meta-Reinforcement Learning","date":"2020-06-12","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-robust-to","title":"Meta-Reinforcement Learning Robust to Distributional Shift via Model Identification and Experience Relabeling","date":"2020-06-12","arxiv_id":"2006.07178","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-model-based-meta-policy-optimization","title":"Meta-Model-Based Meta-Policy Optimization","date":"2020-06-04","arxiv_id":"2006.02608","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-intervention-centric-causal-reasoning","title":"Towards intervention-centric causal reasoning in learning agents","date":"2020-05-26","arxiv_id":"2005.12968","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-trajectory","title":"Meta-Reinforcement Learning for Trajectory Design in Wireless UAV Networks","date":"2020-05-25","arxiv_id":"2005.12394","repositories_listed":0,"syntology":null},{"url":null,"slug":"watch-try-learn-meta-learning-from-1","title":"Watch, Try, Learn: Meta-Learning from Demonstrations and Rewards","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-robotic","title":"Meta-Reinforcement Learning for Robotic Industrial Insertion Tasks","date":"2020-04-29","arxiv_id":"2004.14404","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-context-aware-task-reasoning-for","title":"Learning Context-aware Task Reasoning for Efficient Meta-reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01373","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-meta-reinforcement-learning-for","title":"Multi-Agent Meta-Reinforcement Learning for Self-Powered and Sustainable Edge Computing Systems","date":"2020-02-20","arxiv_id":"2002.08567","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-in-gradient-based-meta","title":"Curriculum in Gradient-Based Meta-Reinforcement Learning","date":"2020-02-19","arxiv_id":"2002.07956","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyper-meta-reinforcement-learning-with-sparse","title":"HMRL: Hyper-Meta Learning for Sparse Reward Reinforcement Learning Problem","date":"2020-02-11","arxiv_id":"2002.04238","repositories_listed":0,"syntology":null},{"url":"/paper/analyzing-policy-distillation-on-multi-task","slug":"analyzing-policy-distillation-on-multi-task","title":"Analyzing Policy Distillation on Multi-Task Learning and Meta-Reinforcement Learning in Meta-World","date":"2020-02-08","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-curricula-for-visual-meta-1","title":"Unsupervised Curricula for Visual Meta-Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.04226","repositories_listed":0,"syntology":null},{"url":null,"slug":"mame-model-agnostic-meta-exploration","title":"MAME : Model-Agnostic Meta-Exploration","date":"2019-11-11","arxiv_id":"1911.04024","repositories_listed":0,"syntology":null},{"url":null,"slug":"state2vec-off-policy-successor-features","title":"State2vec: Off-Policy Successor Features Approximators","date":"2019-10-22","arxiv_id":"1910.10277","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-generalization-in-meta-1","title":"Improving Generalization in Meta Reinforcement Learning using Learned Objectives","date":"2019-10-09","arxiv_id":"1910.04098","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-meta-reinforcement-learning-via-1","title":"MGHRL: Meta Goal-generation for Hierarchical Reinforcement Learning","date":"2019-09-30","arxiv_id":"1909.13607","repositories_listed":0,"syntology":null},{"url":null,"slug":"consistent-meta-reinforcement-learning-via","title":"Consistent Meta-Reinforcement Learning via Model Identification and Experience Relabeling","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-meta-reinforcement-learning-via","title":"Efficient meta reinforcement learning via meta goal generation","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hope-for-the-best-but-prepare-for-the-worst","title":"Hope For The Best But Prepare For The Worst: Cautious Adaptation In RL Agents","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-as-batch-meta-reinforcement","title":"Multi-task Batch Reinforcement Learning with Metric Learning","date":"2019-09-25","arxiv_id":"1909.11373","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-as-batch-meta-reinforcement-1","title":"Pre-training as Batch Meta Reinforcement Learning with tiMe","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-sim-to-real","title":"Meta Reinforcement Learning for Sim-to-real Domain Adaptation","date":"2019-09-16","arxiv_id":"1909.12906","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-adaptation-with-meta-reinforcement","title":"Fast Adaptation with Meta-Reinforcement Learning for Trust Modelling in Human-Robot Interaction","date":"2019-08-12","arxiv_id":"1908.04087","repositories_listed":0,"syntology":null},{"url":null,"slug":"watch-try-learn-meta-learning-from","title":"Watch, Try, Learn: Meta-Learning from Demonstrations and Reward","date":"2019-06-07","arxiv_id":"1906.03352","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-exploration-policies-for-model","title":"Learning Exploration Policies for Model-Agnostic Meta-Reinforcement Learning","date":"2019-05-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-adaptive-1","title":"Meta-Reinforcement Learning for Adaptive Autonomous Driving","date":"2019-05-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-reinforcement-learn-by-imitation","title":"Learning to Reinforcement Learn by Imitation","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-meta-policy-search","title":"Guided Meta-Policy Search","date":"2019-04-01","arxiv_id":"1904.00956","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-with-distribution","title":"Meta Reinforcement Learning with Distribution of Exploration Parameters Learned by Evolution Strategies","date":"2018-12-29","arxiv_id":"1812.11314","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-meta-reinforcement-learning-for-1","title":"A Review of Meta-Reinforcement Learning for Deep Neural Networks Architecture Search","date":"2018-12-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-meta-reinforcement-learning-for","title":"A Review of Meta-Reinforcement Learning for Deep Neural Networks Architecture Search","date":"2018-12-17","arxiv_id":"1812.07995","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-effects-of-negative-adaptation-in-model","title":"The effects of negative adaptation in Model-Agnostic Meta-Learning","date":"2018-12-05","arxiv_id":"1812.02159","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-importance-of-sampling-inmeta","title":"The Importance of Sampling inMeta-Reinforcement Learning","date":"2018-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-meta-learning-for-reinforcement","title":"Unsupervised Meta-Learning for Reinforcement Learning","date":"2018-06-12","arxiv_id":"1806.04640","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-with-latent","title":"Meta Reinforcement Learning with Latent Variable Gaussian Processes","date":"2018-03-20","arxiv_id":"1803.07551","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-episodic-value-iteration-for-model-based","title":"Deep Episodic Value Iteration for Model-based Meta-Reinforcement Learning","date":"2017-05-09","arxiv_id":"1705.03562","repositories_listed":0,"syntology":null}],"record_sha256":"69eb4d70d757a20e1c4c46b4d1de17958382be6bf2f7d36e880cb8eac477dc08","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}