{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/103","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":103,"pages_in_order":152,"rows_per_page":100,"rows":[10201,10300],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/102","next":"/task/reinforcement-learning-1/papers/104","papers":[{"url":null,"slug":"scenic4rl-programmatic-modeling-and","title":"Scenic4RL: Programmatic Modeling and Generation of Reinforcement Learning Environments","date":"2021-06-18","arxiv_id":"2106.10365","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategically-timed-state-observation-attacks","title":"Strategically-timed State-Observation Attacks on Deep Reinforcement Learning Agents","date":"2021-06-18","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach","title":"A Deep Reinforcement Learning Approach towards Pendulum Swing-up Problem based on TF-Agents","date":"2021-06-17","arxiv_id":"2106.09556","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for-an-irs","title":"A Reinforcement Learning Approach for an IRS-assisted NOMA Network","date":"2021-06-17","arxiv_id":"2106.09611","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-the-function-approximation","title":"Adapting the Function Approximation Architecture in Online Reinforcement Learning","date":"2021-06-17","arxiv_id":"2106.09776","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-multi-agent-reinforcement-3","title":"Cooperative Multi-Agent Reinforcement Learning Based Distributed Dynamic Spectrum Access in Cognitive Radio Networks","date":"2021-06-17","arxiv_id":"2106.09274","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-1","title":"Deep Reinforcement Learning Based Optimization for IRS Based UAV-NOMA Downlink Networks","date":"2021-06-17","arxiv_id":"2106.09616","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-automated","title":"Deep reinforcement learning with automated label extraction from clinical reports accurately classifies 3D MRI brain volumes","date":"2021-06-17","arxiv_id":"2106.09812","repositories_listed":0,"syntology":null},{"url":null,"slug":"many-agent-reinforcement-learning-under","title":"Many Agent Reinforcement Learning Under Partial Observability","date":"2021-06-17","arxiv_id":"2106.09825","repositories_listed":0,"syntology":null},{"url":null,"slug":"modelling-resource-allocation-in-uncertain","title":"Modelling resource allocation in uncertain system environment through deep reinforcement learning","date":"2021-06-17","arxiv_id":"2106.09461","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-curricula-via-expert-demonstrations","title":"Automatic Curricula via Expert Demonstrations","date":"2021-06-16","arxiv_id":"2106.09159","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavioral-priors-and-dynamics-models","title":"Behavioral Priors and Dynamics Models: Improving Performance and Domain Transfer in Offline RL","date":"2021-06-16","arxiv_id":"2106.09119","repositories_listed":0,"syntology":null},{"url":null,"slug":"mpc-based-reinforcement-learning-for-a","title":"MPC-based Reinforcement Learning for a Simplified Freight Mission of Autonomous Surface Vehicles","date":"2021-06-16","arxiv_id":"2106.08634","repositories_listed":0,"syntology":null},{"url":null,"slug":"mungojerrie-reinforcement-learning-of-linear","title":"Mungojerrie: Reinforcement Learning of Linear-Time Objectives","date":"2021-06-16","arxiv_id":"2106.09161","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-markovian-bandits","title":"Reinforcement Learning for Markovian Bandits: Is Posterior Sampling more Scalable than Optimism?","date":"2021-06-16","arxiv_id":"2106.08771","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-pursuit-and","title":"Reinforcement learning for pursuit and evasion of microswimmers at low Reynolds number","date":"2021-06-16","arxiv_id":"2106.08609","repositories_listed":0,"syntology":null},{"url":null,"slug":"unbiased-methods-for-multi-goal-reinforcement","title":"Unbiased Methods for Multi-Goal Reinforcement Learning","date":"2021-06-16","arxiv_id":"2106.08863","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-on-a-multi-asset","title":"Deep reinforcement learning on a multi-asset environment for trading","date":"2021-06-15","arxiv_id":"2106.08437","repositories_listed":0,"syntology":null},{"url":null,"slug":"fundamental-limits-of-reinforcement-learning","title":"Fundamental Limits of Reinforcement Learning in Environment with Endogeneous and Exogeneous Uncertainty","date":"2021-06-15","arxiv_id":"2106.08477","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimizing-communication-while-maximizing","title":"Minimizing Communication while Maximizing Performance in Multi-Agent Reinforcement Learning","date":"2021-06-15","arxiv_id":"2106.08482","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-multi-objective-policy-optimization-as-a","title":"On Multi-objective Policy Optimization as a Tool for Reinforcement Learning: Case Studies in Offline RL and Finetuning","date":"2021-06-15","arxiv_id":"2106.08199","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-power-of-multitask-representation","title":"On the Power of Multitask Representation Learning in Linear MDP","date":"2021-06-15","arxiv_id":"2106.08053","repositories_listed":0,"syntology":null},{"url":null,"slug":"population-coding-and-dynamic-neurons","title":"Population-coding and Dynamic-neurons improved Spiking Actor Network for Reinforcement Learning","date":"2021-06-15","arxiv_id":"2106.07854","repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-reinforcement-learning-from","title":"Residual Reinforcement Learning from Demonstrations","date":"2021-06-15","arxiv_id":"2106.08050","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-in-1","title":"Sample Efficient Reinforcement Learning In Continuous State Spaces: A Perspective Beyond Linearity","date":"2021-06-15","arxiv_id":"2106.07814","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-safe-control-of-continuum-manipulator","title":"Towards Safe Control of Continuum Manipulator Using Shielded Multiagent Reinforcement Learning","date":"2021-06-15","arxiv_id":"2106.07892","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-document-sketching-generating","title":"Automatic Document Sketching: Generating Drafts from Analogous Texts","date":"2021-06-14","arxiv_id":"2106.07192","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-aided-heuristics-design-for-storage","title":"Learning-Aided Heuristics Design for Storage System","date":"2021-06-14","arxiv_id":"2106.07288","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-policy-deep-reinforcement-learning-for-the","title":"On-Policy Deep Reinforcement Learning for the Average-Reward Criterion","date":"2021-06-14","arxiv_id":"2106.07329","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-sub-sampling-for-reinforcement","title":"Online Sub-Sampling for Reinforcement Learning with General Function Approximation","date":"2021-06-14","arxiv_id":"2106.07203","repositories_listed":0,"syntology":null},{"url":null,"slug":"poisoning-deep-reinforcement-learning-agents","title":"Poisoning Deep Reinforcement Learning Agents with In-Distribution Triggers","date":"2021-06-14","arxiv_id":"2106.07798","repositories_listed":0,"syntology":null},{"url":null,"slug":"targeted-data-acquisition-for-evolving","title":"Targeted Data Acquisition for Evolving Negotiation Agents","date":"2021-06-14","arxiv_id":"2106.07728","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-like-playing-a-reinforcement","title":"Training like Playing: A Reinforcement Learning And Knowledge Graph-based framework for building Automatic Consultation System in Medical Field","date":"2021-06-14","arxiv_id":"2106.07502","repositories_listed":0,"syntology":null},{"url":null,"slug":"user-guided-personalized-image-aesthetic","title":"User-Guided Personalized Image Aesthetic Assessment based on Deep Reinforcement Learning","date":"2021-06-14","arxiv_id":"2106.07488","repositories_listed":0,"syntology":null},{"url":null,"slug":"which-mutual-information-representation-1","title":"Which Mutual-Information Representation Learning Objectives are Sufficient for Control?","date":"2021-06-14","arxiv_id":"2106.07278","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-soft-computing-method-for-integration","title":"A new soft computing method for integration of expert's knowledge in reinforcement learn-ing problems","date":"2021-06-13","arxiv_id":"2106.07088","repositories_listed":0,"syntology":null},{"url":null,"slug":"bellman-consistent-pessimism-for-offline","title":"Bellman-consistent Pessimism for Offline Reinforcement Learning","date":"2021-06-13","arxiv_id":"2106.06926","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-exploration-with-self-play-for","title":"Data-Efficient Exploration with Self Play for Atari","date":"2021-06-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"density-based-bonuses-on-learned","title":"Density-Based Bonuses on Learned Representations for Reward-Free Exploration in Deep Reinforcement Learning","date":"2021-06-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-predictive-representation-for","title":"Disentangled Predictive Representation for Meta-Reinforcement Learning","date":"2021-06-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-driven-representation-learning-in","title":"Exploration-Driven Representation Learning in Reinforcement Learning","date":"2021-06-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-on-abstract-domains-a-new-approach","title":"Learning on Abstract Domains: A New Approach for Verifiable Guarantee in Reinforcement Learning","date":"2021-06-13","arxiv_id":"2106.06931","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-task-relevant-representations-with","title":"Learning Task-Relevant Representations with Selective Contrast for Reinforcement Learning in a Real-World Application","date":"2021-06-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-explore-multiple-environments","title":"Learning to Explore Multiple Environments without Rewards","date":"2021-06-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"masai-multi-agent-summative-assessment","title":"MASAI: Multi-agent Summative Assessment Improvement for Unsupervised Environment Design","date":"2021-06-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-learning-for-out-of-1","title":"Representation Learning for Out-of-distribution Generalization in Reinforcement Learning","date":"2021-06-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tangent-space-least-adaptive-clustering","title":"Tangent Space Least Adaptive Clustering","date":"2021-06-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-reinforcement-learning-for-2","title":"Model-free Reinforcement Learning for Branching Markov Decision Processes","date":"2021-06-12","arxiv_id":"2106.06777","repositories_listed":0,"syntology":null},{"url":null,"slug":"a3c-s-automated-agent-accelerator-co-search","title":"A3C-S: Automated Agent Accelerator Co-Search towards Efficient Deep Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06577","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-risk-adaptation-in-distributional","title":"Automatic Risk Adaptation in Distributional Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06317","repositories_listed":0,"syntology":null},{"url":null,"slug":"corruption-robust-offline-reinforcement","title":"Corruption-Robust Offline Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06630","repositories_listed":0,"syntology":null},{"url":null,"slug":"courteous-behavior-of-automated-vehicles-at","title":"Courteous Behavior of Automated Vehicles at Unsignalized Intersections via Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06369","repositories_listed":0,"syntology":null},{"url":null,"slug":"decore-deep-compression-with-reinforcement","title":"DECORE: Deep Compression with Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06091","repositories_listed":0,"syntology":null},{"url":"/paper/gdi-rethinking-what-makes-reinforcement","slug":"gdi-rethinking-what-makes-reinforcement","title":"GDI: Rethinking What Makes Reinforcement Learning Different From Supervised Learning","date":"2021-06-11","arxiv_id":"2106.06232","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-as-anti","title":"Offline Reinforcement Learning as Anti-Exploration","date":"2021-06-11","arxiv_id":"2106.06431","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-linear","title":"Safe Reinforcement Learning with Linear Function Approximation","date":"2021-06-11","arxiv_id":"2106.06239","repositories_listed":0,"syntology":null},{"url":null,"slug":"taylor-expansion-of-discount-factors","title":"Taylor Expansion of Discount Factors","date":"2021-06-11","arxiv_id":"2106.06170","repositories_listed":0,"syntology":null},{"url":null,"slug":"to-beam-or-not-to-beam-that-is-a-question-of","title":"To Beam Or Not To Beam: That is a Question of Cooperation for Language GANs","date":"2021-06-11","arxiv_id":"2106.06363","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-battery-operation-for-energy","title":"Data-driven battery operation for energy arbitrage using rainbow deep reinforcement learning","date":"2021-06-10","arxiv_id":"2106.06061","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperspace-neighbor-penetration-approach-to","title":"Hyperspace Neighbor Penetration Approach to Dynamic Programming for Model-Based Reinforcement Learning Problems with Slowly Changing Variables in A Continuous State Space","date":"2021-06-10","arxiv_id":"2106.05497","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlcorrector-reinforced-proofreading-for","title":"RLCorrector: Reinforced Proofreading for Cell-level Microscopy Image Segmentation","date":"2021-06-10","arxiv_id":"2106.05487","repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-with-a-mixture-prior","title":"Thompson Sampling with a Mixture Prior","date":"2021-06-10","arxiv_id":"2106.05608","repositories_listed":0,"syntology":null},{"url":null,"slug":"5g-mimo-data-for-machine-learning-application","title":"5G MIMO Data for Machine Learning: Application to Beam-Selection using Deep Learning","date":"2021-06-09","arxiv_id":"2106.05370","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-bellman-operators","title":"Bayesian Bellman Operators","date":"2021-06-09","arxiv_id":"2106.05012","repositories_listed":0,"syntology":null},{"url":null,"slug":"deception-in-social-learning-a-multi-agent","title":"Deception in Social Learning: A Multi-Agent Reinforcement Learning Perspective","date":"2021-06-09","arxiv_id":"2106.05402","repositories_listed":0,"syntology":null},{"url":null,"slug":"eye-of-the-beholder-improved-relation","title":"Eye of the Beholder: Improved Relation Generalization for Text-based Reinforcement Learning Agents","date":"2021-06-09","arxiv_id":"2106.05387","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-inverse-reinforcement-learning","title":"Offline Inverse Reinforcement Learning","date":"2021-06-09","arxiv_id":"2106.05068","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-learning-for-stochastic-shortest-path","title":"Online Learning for Stochastic Shortest Path Model via Posterior Sampling","date":"2021-06-09","arxiv_id":"2106.05335","repositories_listed":0,"syntology":null},{"url":null,"slug":"over-the-fiber-digital-predistortion-using","title":"Over-the-fiber Digital Predistortion Using Reinforcement Learning","date":"2021-06-09","arxiv_id":"2106.04934","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-finetuning-bridging-sample-efficient","title":"Policy Finetuning: Bridging Sample-Efficient Offline and Online Reinforcement Learning","date":"2021-06-09","arxiv_id":"2106.04895","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-industrial-control","title":"Reinforcement Learning for Industrial Control Network Cyber Security Orchestration","date":"2021-06-09","arxiv_id":"2106.05332","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-value-network-based-approach-for-multi","title":"A Deep Value-network Based Approach for Multi-Driver Order Dispatching","date":"2021-06-08","arxiv_id":"2106.04493","repositories_listed":0,"syntology":null},{"url":null,"slug":"don-t-get-yourself-into-trouble-risk-aware","title":"Don't Get Yourself into Trouble! Risk-aware Decision-Making for Autonomous Vehicles","date":"2021-06-08","arxiv_id":"2106.04625","repositories_listed":0,"syntology":null},{"url":null,"slug":"left-ventricle-contouring-in-cardiac-images","title":"Left Ventricle Contouring in Cardiac Images Based on Deep Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04127","repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-feedback-learning-for-contact-rich","title":"Residual Feedback Learning for Contact-Rich Manipulation Tasks with Uncertainty","date":"2021-06-08","arxiv_id":"2106.04306","repositories_listed":0,"syntology":null},{"url":null,"slug":"rewardsofsum-exploring-reinforcement-learning","title":"RewardsOfSum: Exploring Reinforcement Learning Rewards for Summarisation","date":"2021-06-08","arxiv_id":"2106.04080","repositories_listed":0,"syntology":null},{"url":null,"slug":"there-is-no-turning-back-a-self-supervised","title":"There Is No Turning Back: A Self-Supervised Approach for Reversibility-Aware Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04480","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-practical-credit-assignment-for-deep","title":"Towards Practical Credit Assignment for Deep Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04499","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-computational-model-of-representation","title":"A Computational Model of Representation Learning in the Brain Cortex, Integrating Unsupervised and Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03688","repositories_listed":0,"syntology":null},{"url":null,"slug":"average-reward-reinforcement-learning-with-2","title":"Average-Reward Reinforcement Learning with Trust Region Methods","date":"2021-06-07","arxiv_id":"2106.03442","repositories_listed":0,"syntology":null},{"url":null,"slug":"concave-utility-reinforcement-learning-the","title":"Concave Utility Reinforcement Learning: the Mean-Field Game Viewpoint","date":"2021-06-07","arxiv_id":"2106.03787","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-artificial-intelligence-xai-for-1","title":"Explainable Artificial Intelligence (XAI) for Increasing User Trust in Deep Reinforcement Learning Driven Autonomous Systems","date":"2021-06-07","arxiv_id":"2106.03775","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifiability-in-inverse-reinforcement","title":"Identifiability in inverse reinforcement learning","date":"2021-06-07","arxiv_id":"2106.03498","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-combinatorial-node-labeling","title":"Learning Combinatorial Node Labeling Algorithms","date":"2021-06-07","arxiv_id":"2106.03594","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-guide-a-saturation-based-theorem","title":"Learning to Guide a Saturation-Based Theorem Prover","date":"2021-06-07","arxiv_id":"2106.03906","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-without-knowing-unobserved-context","title":"Learning without Knowing: Unobserved Context in Continuous Transfer Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03833","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-battery-storage-management-using","title":"Multi-agent Battery Storage Management using MPC-based Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03541","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-power-of-exploiter-provable-multi-agent","title":"The Power of Exploiter: Provable Multi-Agent RL in Large State Spaces","date":"2021-06-07","arxiv_id":"2106.03352","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-and-domain-agnostic","title":"Towards robust and domain agnostic reinforcement learning competitions","date":"2021-06-07","arxiv_id":"2106.03748","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-uav-trajectory-and-data-collection","title":"3D UAV Trajectory and Data Collection Optimisation via Deep Reinforcement Learning","date":"2021-06-06","arxiv_id":"2106.03129","repositories_listed":0,"syntology":null},{"url":null,"slug":"distop-discovering-a-topological","title":"DisTop: Discovering a Topological representation to learn diverse and rewarding skills","date":"2021-06-06","arxiv_id":"2106.03853","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-mdps-from-features-predict-then","title":"Learning MDPs from Features: Predict-Then-Optimize for Sequential Decision Problems by Reinforcement Learning","date":"2021-06-06","arxiv_id":"2106.03279","repositories_listed":0,"syntology":null},{"url":null,"slug":"heuristic-guided-reinforcement-learning","title":"Heuristic-Guided Reinforcement Learning","date":"2021-06-05","arxiv_id":"2106.02757","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-routines-for-effective-off-policy","title":"Learning Routines for Effective Off-Policy Reinforcement Learning","date":"2021-06-05","arxiv_id":"2106.02943","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-assignment-problem-1","title":"Reinforcement Learning for Assignment Problem with Time Constraints","date":"2021-06-05","arxiv_id":"2106.02856","repositories_listed":0,"syntology":null},{"url":null,"slug":"be-considerate-objectives-side-effects-and","title":"Be Considerate: Objectives, Side Effects, and Deciding How to Act","date":"2021-06-04","arxiv_id":"2106.02617","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-and-adapting-to-novelty-in-games","title":"Detecting and Adapting to Novelty in Games","date":"2021-06-04","arxiv_id":"2106.02204","repositories_listed":0,"syntology":null},{"url":null,"slug":"nara-learning-network-aware-resource","title":"Resource Allocation in Disaggregated Data Centre Systems with Reinforcement Learning","date":"2021-06-04","arxiv_id":"2106.02412","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustifying-reinforcement-learning-policies","title":"Robustifying Reinforcement Learning Policies with $\\mathcal{L}_1$ Adaptive Control","date":"2021-06-04","arxiv_id":"2106.02249","repositories_listed":0,"syntology":null},{"url":null,"slug":"feeling-of-presence-maximization-mmwave","title":"Feeling of Presence Maximization: mmWave-Enabled Virtual Reality Meets Deep Reinforcement Learning","date":"2021-06-03","arxiv_id":"2107.01001","repositories_listed":0,"syntology":null}],"record_sha256":"295188c4c1325f4ba5b1a4aa8e9a857438ed24a78dd38539ef296893b5845cf7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}