{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/meta-reinforcement-learning/papers/2","list_of":"/task/meta-reinforcement-learning","task":"Meta Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":3,"rows_per_page":100,"rows":[101,200],"of":278,"counts":{"archive_papers_tagged":278,"with_a_code_link":103,"where_syntology_ran_a_sample":39,"not_listed_spam_title":0,"listed":278,"listed_where_code_ran":39,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":32,"every_run_a_failure_of_syntologys_instrument":7,"listed_with_a_run_with_no_instrument_failure":32,"listed_every_run_a_failure_of_syntologys_instrument":7,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/meta-reinforcement-learning","prev":"/task/meta-reinforcement-learning","next":"/task/meta-reinforcement-learning/papers/3","papers":[{"url":"/paper/concurrent-meta-reinforcement-learning","slug":"concurrent-meta-reinforcement-learning","title":"Concurrent Meta Reinforcement Learning","date":"2019-03-07","arxiv_id":"1903.02710","repositories_listed":1,"syntology":null},{"url":"/paper/causal-reasoning-from-meta-reinforcement","slug":"causal-reasoning-from-meta-reinforcement","title":"Causal Reasoning from Meta-reinforcement Learning","date":"2019-01-23","arxiv_id":"1901.08162","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/causal-reasoning-from-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1901.08162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08162"}},"official":null}},{"url":"/paper/introducing-neuromodulation-in-deep-neural","slug":"introducing-neuromodulation-in-deep-neural","title":"Introducing Neuromodulation in Deep Neural Networks to Learn Adaptive Behaviours","date":"2018-12-21","arxiv_id":"1812.09113","repositories_listed":1,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-fast-and-data","title":"Meta-Reinforcement Learning for Fast and Data-Efficient Spectrum Allocation in Dynamic Wireless Networks","date":"2025-07-13","arxiv_id":"2507.10619","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-algorithm-distillation-for-continuous","title":"Scaling Algorithm Distillation for Continuous Control with Mamba","date":"2025-06-16","arxiv_id":"2506.13892","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-meta-testing-with-conditional","title":"Unsupervised Meta-Testing with Conditional Neural Processes for Hybrid Meta-Reinforcement Learning","date":"2025-06-04","arxiv_id":"2506.04399","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-with-minimum","title":"Meta-reinforcement learning with minimum attention","date":"2025-05-22","arxiv_id":"2505.16741","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-and-robust-task-sampling-with-posterior","title":"Fast and Robust: Task Sampling with Posterior and Diversity Synergies for Adaptive Decision-Makers in Randomized Environments","date":"2025-04-27","arxiv_id":"2504.19139","repositories_listed":0,"syntology":null},{"url":null,"slug":"instructrag-leveraging-retrieval-augmented","title":"InstructRAG: Leveraging Retrieval-Augmented Generation on Instruction Graphs for LLM-Based Task Planning","date":"2025-04-17","arxiv_id":"2504.13032","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-world-models-emerge-from","title":"Embodied World Models Emerge from Navigational Task in Open-Ended Environments","date":"2025-04-15","arxiv_id":"2504.11419","repositories_listed":0,"syntology":null},{"url":null,"slug":"uas-visual-navigation-in-large-and-unseen","title":"UAS Visual Navigation in Large and Unseen Environments via a Meta Agent","date":"2025-03-20","arxiv_id":"2503.15781","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-with-discrete","title":"Meta-Reinforcement Learning with Discrete World Models for Adaptive Load Balancing","date":"2025-03-11","arxiv_id":"2503.08872","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-test-time-compute-via-meta","title":"Optimizing Test-Time Compute via Meta Reinforcement Fine-Tuning","date":"2025-03-10","arxiv_id":"2503.07572","repositories_listed":0,"syntology":null},{"url":null,"slug":"teleology-driven-affective-computing-a-causal","title":"Teleology-Driven Affective Computing: A Causal Framework for Sustained Well-Being","date":"2025-02-24","arxiv_id":"2502.17172","repositories_listed":0,"syntology":null},{"url":"/paper/prism-a-robust-framework-for-skill-based-meta","slug":"prism-a-robust-framework-for-skill-based-meta","title":"PRISM: A Robust Framework for Skill-based Meta-Reinforcement Learning with Noisy Demonstrations","date":"2025-02-06","arxiv_id":"2502.03752","repositories_listed":0,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prism-a-robust-framework-for-skill-based-meta#ran","syntology_url":"https://syntology.ai/paper/2502.03752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.03752"}},"official":null}},{"url":null,"slug":"toward-task-generalization-via-memory","title":"Toward Task Generalization via Memory Augmentation in Meta-Reinforcement Learning","date":"2025-02-03","arxiv_id":"2502.01521","repositories_listed":0,"syntology":null},{"url":null,"slug":"timrl-a-novel-meta-reinforcement-learning","title":"TIMRL: A Novel Meta-Reinforcement Learning Framework for Non-Stationary and Multi-Task Environments","date":"2025-01-13","arxiv_id":"2501.07146","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-multi-agent-meta-reinforcement","title":"Hierarchical Multi-agent Meta-Reinforcement Learning for Cross-channel Bidding","date":"2024-12-26","arxiv_id":"2412.19064","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-meta-reinforcement-learning-via","title":"Hierarchical Meta-Reinforcement Learning via Automated Macro-Action Discovery","date":"2024-12-16","arxiv_id":"2412.11930","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergence-of-implicit-world-models-from","title":"Emergence of Implicit World Models from Mortal Agents","date":"2024-11-19","arxiv_id":"2411.12304","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-with-universal","title":"Meta-Reinforcement Learning with Universal Policy Adaptation: Provable Near-Optimality under All-task Optimum Comparator","date":"2024-10-13","arxiv_id":"2410.09728","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-approach-for","title":"Meta Reinforcement Learning Approach for Adaptive Resource Optimization in O-RAN","date":"2024-09-30","arxiv_id":"2410.03737","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-truly-massive-budgeted-monotonic","title":"Solving Truly Massive Budgeted Monotonic POMDPs with Oracle-Guided Meta-Reinforcement Learning","date":"2024-08-13","arxiv_id":"2408.07192","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-sampling-guided-meta-training-for","title":"Importance Sampling-Guided Meta-Training for Intelligent Agents in Highly Interactive Environments","date":"2024-07-22","arxiv_id":"2407.15839","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-meta-agnostic-reinforcement","title":"Constrained Meta Agnostic Reinforcement Learning","date":"2024-06-20","arxiv_id":"2406.14047","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-sequence-length-of-data-sampling","title":"Memory Sequence Length of Data Sampling Impacts the Adaptation of Meta-Reinforcement Learning Agents","date":"2024-06-18","arxiv_id":"2406.12359","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-time-regret-minimization-in-meta","title":"Test-Time Regret Minimization in Meta Reinforcement Learning","date":"2024-06-04","arxiv_id":"2406.02282","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cmdp-within-online-framework-for-meta-safe","title":"A CMDP-within-online framework for Meta-Safe Reinforcement Learning","date":"2024-05-26","arxiv_id":"2405.16601","repositories_listed":0,"syntology":null},{"url":null,"slug":"theoretical-analysis-of-meta-reinforcement","title":"Theoretical Analysis of Meta Reinforcement Learning: Generalization Bounds and Convergence Guarantees","date":"2024-05-22","arxiv_id":"2405.13290","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-resource-1","title":"Meta Reinforcement Learning for Resource Allocation in Multi-Antenna UAV Network with Rate Splitting Multiple Access","date":"2024-05-18","arxiv_id":"2405.11306","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-resource-2","title":"On the Performance of Unmanned Aerial Vehicles with MIMO VLC","date":"2024-05-18","arxiv_id":"2405.11161","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamite-rl-a-dynamic-model-for-improved","title":"DynaMITE-RL: A Dynamic Model for Improved Temporal Meta-Reinforcement Learning","date":"2024-02-25","arxiv_id":"2402.15957","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-transformers-are-efficient-meta","title":"Hierarchical Transformers are Efficient Meta-Reinforcement Learners","date":"2024-02-09","arxiv_id":"2402.06402","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysing-the-sample-complexity-of-opponent","title":"Analysing the Sample Complexity of Opponent Shaping","date":"2024-02-08","arxiv_id":"2402.05782","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-context-learning-agents-are-asymmetric","title":"In-context learning agents are asymmetric belief updaters","date":"2024-02-06","arxiv_id":"2402.03969","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-abstract-visuomotor-mappings","title":"Learning to Abstract Visuomotor Mappings using Meta-Reinforcement Learning","date":"2024-02-05","arxiv_id":"2402.03072","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-strategic-iot","title":"Meta Reinforcement Learning for Strategic IoT Deployments Coverage in Disaster-Response UAV Swarms","date":"2024-01-20","arxiv_id":"2401.11118","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-an-adaptable-and-generalizable","title":"Towards an Adaptable and Generalizable Optimization Engine in Decision and Control: A Meta Reinforcement Learning Approach","date":"2024-01-04","arxiv_id":"2401.02508","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-multi-task","title":"Meta Reinforcement Learning for Multi-Task Offloading in Vehicular Edge Computing","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-task-relevant-loss-functions-in-meta","title":"On Task-Relevant Loss Functions in Meta-Reinforcement Learning and Online LQR","date":"2023-12-09","arxiv_id":"2312.05465","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-agents-and-data-quality-in-agent","title":"Adaptive Agents and Data Quality in Agent-Based Financial Markets","date":"2023-11-27","arxiv_id":"2311.15974","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-mrl-based-design-solution-for-ris-assisted","title":"An MRL-Based Design Solution for RIS-Assisted MU-MIMO Wireless System under Time-Varying Channels","date":"2023-11-15","arxiv_id":"2311.08840","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-introduction-to-reinforcement-learning-for","title":"An introduction to reinforcement learning for neuroscience","date":"2023-11-13","arxiv_id":"2311.07315","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-task-generalization-via","title":"Data-Efficient Task Generalization via Probabilistic Model-based Meta Reinforcement Learning","date":"2023-11-13","arxiv_id":"2311.07558","repositories_listed":0,"syntology":null},{"url":null,"slug":"dream-to-adapt-meta-reinforcement-learning-by","title":"Dream to Adapt: Meta Reinforcement Learning by Latent Context Imagination and MDP Imagination","date":"2023-11-11","arxiv_id":"2311.06673","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypothesis-network-planned-exploration-for","title":"Hypothesis Network Planned Exploration for Rapid Meta-Reinforcement Learning Adaptation","date":"2023-11-07","arxiv_id":"2311.03701","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergence-of-collective-open-ended","title":"Emergence of Collective Open-Ended Exploration from Decentralized Meta-Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00651","repositories_listed":0,"syntology":null},{"url":null,"slug":"neurosymbolic-meta-reinforcement-lookahead","title":"Neurosymbolic Meta-Reinforcement Lookahead Learning Achieves Safe Self-Driving in Non-Stationary Environments","date":"2023-09-05","arxiv_id":"2309.02328","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-driving-policy-learning-with-guided","title":"Robust Driving Policy Learning with Guided Meta Reinforcement Learning","date":"2023-07-19","arxiv_id":"2307.10160","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-generative-flow-networks-with","title":"Meta Generative Flow Networks with Personalization for Task-Specific Adaptation","date":"2023-06-16","arxiv_id":"2306.09742","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-embodied-language-learning-as-a","title":"Simple Embodied Language Learning as a Byproduct of Meta-Reinforcement Learning","date":"2023-06-14","arxiv_id":"2306.08400","repositories_listed":0,"syntology":null},{"url":null,"slug":"stepsize-learning-for-policy-gradient-methods","title":"Stepsize Learning for Policy Gradient Methods in Contextual Markov Decision Processes","date":"2023-06-13","arxiv_id":"2306.07741","repositories_listed":0,"syntology":null},{"url":null,"slug":"doing-the-right-thing-for-the-right-reason","title":"Doing the right thing for the right reason: Evaluating artificial moral cognition by probing cost insensitivity","date":"2023-05-29","arxiv_id":"2305.18269","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-first-order-meta-reinforcement-learning","title":"On First-Order Meta-Reinforcement Learning with Moreau Envelopes","date":"2023-05-20","arxiv_id":"2305.12216","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-based-on-self","title":"Meta-Reinforcement Learning Based on Self-Supervised Task Representation Learning","date":"2023-04-29","arxiv_id":"2305.00286","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-via-exploratory","title":"Meta-Reinforcement Learning via Exploratory Task Clustering","date":"2023-02-15","arxiv_id":"2302.07958","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-meta-reinforcement-learning","title":"A Survey of Meta-Reinforcement Learning","date":"2023-01-19","arxiv_id":"2301.08028","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-timescale-adaptation-in-an-open-ended","title":"Human-Timescale Adaptation in an Open-Ended Task Space","date":"2023-01-18","arxiv_id":"2301.07608","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuro-symbolic-meta-reinforcement-learning","title":"Neuro-symbolic Meta Reinforcement Learning for Trading","date":"2023-01-15","arxiv_id":"2302.08996","repositories_listed":0,"syntology":null},{"url":null,"slug":"pomrl-no-regret-learning-to-plan-with","title":"POMRL: No-Regret Learning-to-Plan with Increasing Horizons","date":"2022-12-30","arxiv_id":"2212.14530","repositories_listed":0,"syntology":null},{"url":null,"slug":"level-k-meta-learning-for-pedestrian-aware","title":"Cognitive Level-$k$ Meta-Learning for Safe and Pedestrian-Aware Autonomous Driving","date":"2022-12-17","arxiv_id":"2212.08800","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-learning-using-adaptable-task-based","title":"Active learning using adaptable task-based prioritisation","date":"2022-12-03","arxiv_id":"2212.01703","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-transformer-for-offline-meta","title":"Contextual Transformer for Offline Meta Reinforcement Learning","date":"2022-11-15","arxiv_id":"2211.08016","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-using-model","title":"Meta-Reinforcement Learning Using Model Parameters","date":"2022-10-27","arxiv_id":"2210.15515","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-based-meta-reinforcement-learning","title":"Uncertainty-based Meta-Reinforcement Learning for Robust Radar Tracking","date":"2022-10-26","arxiv_id":"2210.14532","repositories_listed":0,"syntology":null},{"url":null,"slug":"metaems-a-meta-reinforcement-learning-based","title":"MetaEMS: A Meta Reinforcement Learning-based Control Framework for Building Energy Management System","date":"2022-10-23","arxiv_id":"2210.12590","repositories_listed":0,"syntology":null},{"url":"/paper/distributionally-adaptive-meta-reinforcement","slug":"distributionally-adaptive-meta-reinforcement","title":"Distributionally Adaptive Meta Reinforcement Learning","date":"2022-10-06","arxiv_id":"2210.03104","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/distributionally-adaptive-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.03104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03104"}},"official":null}},{"url":null,"slug":"meta-reinforcement-learning-for-optimal","title":"Meta Reinforcement Learning for Optimal Design of Legged Robots","date":"2022-10-06","arxiv_id":"2210.02750","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-policy-transfer-with-disentangled-1","title":"Zero-Shot Policy Transfer with Disentangled Task Representation of Meta-Reinforcement Learning","date":"2022-10-01","arxiv_id":"2210.00350","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-symmetry-meta-reinforcement","title":"Learning from Symmetry: Meta-Reinforcement Learning with Symmetrical Behaviors and Language Instructions","date":"2022-09-21","arxiv_id":"2209.10656","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-convergence-theory-of-meta","title":"On the Convergence Theory of Meta Reinforcement Learning with Personalized Policies","date":"2022-09-21","arxiv_id":"2209.10072","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-adaptive-3","title":"Meta-Reinforcement Learning for Adaptive Control of Second Order Systems","date":"2022-09-19","arxiv_id":"2209.09301","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-biological-sequences-via-meta","title":"Designing Biological Sequences via Meta-Reinforcement Learning and Bayesian Optimization","date":"2022-09-13","arxiv_id":"2209.06259","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-model-based-approach-to-meta-reinforcement","title":"A model-based approach to meta-Reinforcement Learning: Transformers and tree search","date":"2022-08-24","arxiv_id":"2208.11535","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-multi-agent-meta-reinforcement","title":"Quantum Multi-Agent Meta Reinforcement Learning","date":"2022-08-22","arxiv_id":"2208.11510","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-with-successor","title":"Meta Reinforcement Learning with Successor Feature Based Context","date":"2022-07-29","arxiv_id":"2207.14723","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-action-translator-for-meta","title":"Learning Action Translator for Meta Reinforcement Learning on Sparse-Reward Tasks","date":"2022-07-19","arxiv_id":"2207.09071","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-meta-reinforcement-learning-for-uav","title":"Continual Meta-Reinforcement Learning for UAV-Aided Vehicular Wireless Networks","date":"2022-07-13","arxiv_id":"2207.06131","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-decision-transformer-for-few-shot","title":"Prompting Decision Transformer for Few-Shot Policy Generalization","date":"2022-06-27","arxiv_id":"2206.13499","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-and-optimal-adaptive-tracking-control-a","title":"Fast and Optimal Adaptive Tracking Control: A Novel Meta-Reinforcement Learning via Conditional Generative Adversarial Net","date":"2022-06-24","arxiv_id":"2206.12450","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effectiveness-of-fine-tuning-versus","title":"On the Effectiveness of Fine-tuning Versus Meta-reinforcement Learning","date":"2022-06-07","arxiv_id":"2206.03271","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-meta-reinforcement-learning-for","title":"Variational Meta Reinforcement Learning for Social Robotics","date":"2022-06-07","arxiv_id":"2206.03211","repositories_listed":0,"syntology":null},{"url":null,"slug":"gramer-graph-meta-reinforcement-learning-for","title":"GraMeR: Graph Meta Reinforcement Learning for Multi-Objective Influence Maximization","date":"2022-05-30","arxiv_id":"2205.14834","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-inference-and-transfer-of-compositional","title":"Fast Inference and Transfer of Compositional Task Structures for Few-shot Task Generalization","date":"2022-05-25","arxiv_id":"2205.12648","repositories_listed":0,"syntology":null},{"url":null,"slug":"skill-based-meta-reinforcement-learning-1","title":"Skill-based Meta-Reinforcement Learning","date":"2022-04-25","arxiv_id":"2204.11828","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-meta-reinforcement-learning-with","title":"Robust Meta-Reinforcement Learning with Curriculum-Based Task Sampling","date":"2022-03-31","arxiv_id":"2203.16801","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-adaptive-2","title":"Meta-Reinforcement Learning for the Tuning of PI Controllers: An Offline Approach","date":"2022-03-17","arxiv_id":"2203.09661","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-offline-meta-reinforcement-1","title":"Model-Based Offline Meta-Reinforcement Learning with Regularization","date":"2022-02-07","arxiv_id":"2202.02929","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-discourse-on-metods-meta-optimized","title":"Meta-Reinforcement Learning with Self-Modifying Networks","date":"2022-02-04","arxiv_id":"2202.02363","repositories_listed":0,"syntology":null},{"url":null,"slug":"reldec-reinforcement-learning-based-decoding","title":"RELDEC: Reinforcement Learning-Based Decoding of Moderate Length LDPC Codes","date":"2021-12-27","arxiv_id":"2112.13934","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-channel-access-via-meta-reinforcement","title":"Dynamic Channel Access via Meta-Reinforcement Learning","date":"2021-12-24","arxiv_id":"2201.09075","repositories_listed":0,"syntology":null},{"url":null,"slug":"biased-gradient-estimate-with-drastic","title":"Biased Gradient Estimate with Drastic Variance Reduction for Meta Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07328","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-cpr-generalize-to-unseen-large-number-of","title":"Meta-CPR: Generalize to Unseen Large Number of Agents with Communication Pattern Recognition Module","date":"2021-12-14","arxiv_id":"2112.07222","repositories_listed":0,"syntology":null},{"url":null,"slug":"comps-continual-meta-policy-search-1","title":"CoMPS: Continual Meta Policy Search","date":"2021-12-08","arxiv_id":"2112.04467","repositories_listed":0,"syntology":null},{"url":null,"slug":"hindsight-task-relabelling-experience-replay-1","title":"Hindsight Task Relabelling: Experience Replay for Sparse Reward Meta-RL","date":"2021-12-02","arxiv_id":"2112.00901","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-practical-consistency-of-meta","title":"On the Practical Consistency of Meta-Reinforcement Learning Algorithms","date":"2021-12-01","arxiv_id":"2112.00478","repositories_listed":0,"syntology":null},{"url":null,"slug":"mamrl-exploiting-multi-agent-meta","title":"MAMRL: Exploiting Multi-agent Meta Reinforcement Learning in WAN Traffic Engineering","date":"2021-11-30","arxiv_id":"2111.15087","repositories_listed":0,"syntology":null},{"url":null,"slug":"modellight-model-based-meta-reinforcement","title":"ModelLight: Model-Based Meta-Reinforcement Learning for Traffic Signal Control","date":"2021-11-15","arxiv_id":"2111.08067","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-cooperate-with-unseen-agent-via","title":"Learning to Cooperate with Unseen Agent via Meta-Reinforcement Learning","date":"2021-11-05","arxiv_id":"2111.03431","repositories_listed":0,"syntology":null},{"url":null,"slug":"alphad3m-machine-learning-pipeline-synthesis","title":"AlphaD3M: Machine Learning Pipeline Synthesis","date":"2021-11-03","arxiv_id":"2111.02508","repositories_listed":0,"syntology":null}],"record_sha256":"a6040f85f5853dc59e98607403c8aabb6df051310407989fa04e2736e12965c3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}