{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/82","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":82,"pages_in_order":135,"rows_per_page":100,"rows":[8101,8200],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/81","next":"/task/reinforcement-learning-2/papers/83","papers":[{"url":null,"slug":"external-control-of-a-genetic-toggle-switch","title":"External control of a genetic toggle switch via Reinforcement Learning","date":"2022-04-11","arxiv_id":"2204.04972","repositories_listed":0,"syntology":null},{"url":null,"slug":"implementing-online-reinforcement-learning","title":"Implementing Online Reinforcement Learning with Temporal Neural Networks","date":"2022-04-11","arxiv_id":"2204.05437","repositories_listed":0,"syntology":null},{"url":null,"slug":"settling-the-sample-complexity-of-model-based","title":"Settling the Sample Complexity of Model-Based Offline Reinforcement Learning","date":"2022-04-11","arxiv_id":"2204.05275","repositories_listed":0,"syntology":null},{"url":null,"slug":"neurl-closed-form-inverse-reinforcement-1","title":"NeuRL: Closed-form Inverse Reinforcement Learning for Neural Decoding","date":"2022-04-10","arxiv_id":"2204.04733","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-spiking-neural-network-structure","title":"A Spiking Neural Network Structure Implementing Reinforcement Learning","date":"2022-04-09","arxiv_id":"2204.04431","repositories_listed":0,"syntology":null},{"url":null,"slug":"hardware-trojan-insertion-using-reinforcement","title":"Hardware Trojan Insertion Using Reinforcement Learning","date":"2022-04-09","arxiv_id":"2204.04350","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-discounting-free-policy","title":"Approximate discounting-free policy evaluation from transient and recurrent states","date":"2022-04-08","arxiv_id":"2204.04324","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-evaluation-of-training-action-1","title":"Data-Driven Evaluation of Training Action Space for Reinforcement Learning","date":"2022-04-08","arxiv_id":"2204.03840","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-complexity-of-markov-equilibrium-in","title":"The Complexity of Markov Equilibrium in Stochastic Games","date":"2022-04-08","arxiv_id":"2204.03991","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-reinforcement-learning-for-robot","title":"Distributed Reinforcement Learning for Robot Teams: A Review","date":"2022-04-07","arxiv_id":"2204.03516","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitating-fast-and-slow-robust-learning-from","title":"Imitating, Fast and Slow: Robust learning from demonstrations via decision-time planning","date":"2022-04-07","arxiv_id":"2204.03597","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-with-online-random-forests","title":"Q-learning with online random forests","date":"2022-04-07","arxiv_id":"2204.03771","repositories_listed":0,"syntology":null},{"url":null,"slug":"standardized-feature-extraction-from-pairwise","title":"Standardized feature extraction from pairwise conflicts applied to the train rescheduling problem","date":"2022-04-06","arxiv_id":"2204.03061","repositories_listed":0,"syntology":null},{"url":null,"slug":"configuration-path-control","title":"Configuration Path Control","date":"2022-04-05","arxiv_id":"2204.02471","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-computational-consequences-of-cost","title":"On the Computational Consequences of Cost Function Design in Nonlinear Optimal Control","date":"2022-04-05","arxiv_id":"2204.01986","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl4real-reinforcement-learning-for-register","title":"RL4ReAl: Reinforcement Learning for Register Allocation","date":"2022-04-05","arxiv_id":"2204.02013","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimising-energy-efficiency-in-uav-assisted","title":"Optimising Energy Efficiency in UAV-Assisted Networks using Deep Reinforcement Learning","date":"2022-04-04","arxiv_id":"2204.01597","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-controller-for-output-feedback-linear","title":"Safe Controller for Output Feedback Linear Systems using Model-Based Reinforcement Learning","date":"2022-04-04","arxiv_id":"2204.01409","repositories_listed":0,"syntology":null},{"url":null,"slug":"best-response-bayesian-reinforcement-learning","title":"Best-Response Bayesian Reinforcement Learning with Bayes-adaptive POMDPs for Centaurs","date":"2022-04-03","arxiv_id":"2204.01160","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-physical-activity-recommendation-on","title":"Enhancing Digital Health Services: A Machine Learning Approach to Personalized Exercise Goal Setting","date":"2022-04-03","arxiv_id":"2204.00961","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-data-aided-channel-estimation-for-mimo","title":"Semi-Data-Aided Channel Estimation for MIMO Systems via Reinforcement Learning","date":"2022-04-03","arxiv_id":"2204.01052","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-transfer-in-deep-reinforcement","title":"Hybrid Transfer in Deep Reinforcement Learning for Ads Allocation","date":"2022-04-02","arxiv_id":"2204.11589","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-via-shielding-for","title":"Safe Reinforcement Learning via Shielding under Partial Observability","date":"2022-04-02","arxiv_id":"2204.00755","repositories_listed":0,"syntology":null},{"url":null,"slug":"automating-staged-rollout-with-reinforcement","title":"Automating Staged Rollout with Reinforcement Learning","date":"2022-04-01","arxiv_id":"2204.02189","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-decision-forest-via-deep","title":"Building Decision Forest via Deep Reinforcement Learning","date":"2022-04-01","arxiv_id":"2204.00306","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-page-level-interest-network-in","title":"Deep Page-Level Interest Network in Reinforcement Learning for Ads Allocation","date":"2022-04-01","arxiv_id":"2204.00377","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-agnostic-counterfactual-synthesis","title":"Model-agnostic Counterfactual Synthesis Policy for Interactive Recommendation","date":"2022-04-01","arxiv_id":"2204.00308","repositories_listed":0,"syntology":null},{"url":null,"slug":"cogngen-constructing-the-kernel-of-a","title":"Maze Learning using a Hyperdimensional Predictive Processing Cognitive Architecture","date":"2022-03-31","arxiv_id":"2204.00619","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-graph","title":"Spatio-Temporal Graph Convolutional Neural Networks for Physics-Aware Grid Learning Algorithms","date":"2022-03-31","arxiv_id":"2203.16732","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffskill-skill-abstraction-from-1","title":"DiffSkill: Skill Abstraction from Differentiable Physics for Deformable Object Manipulations with Tools","date":"2022-03-31","arxiv_id":"2203.17275","repositories_listed":0,"syntology":null},{"url":null,"slug":"mask-atari-for-deep-reinforcement-learning-as","title":"Mask Atari for Deep Reinforcement Learning as POMDP Benchmarks","date":"2022-03-31","arxiv_id":"2203.16777","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-meta-reinforcement-learning-with","title":"Robust Meta-Reinforcement Learning with Curriculum-Based Task Sampling","date":"2022-03-31","arxiv_id":"2203.16801","repositories_listed":0,"syntology":null},{"url":null,"slug":"trajgen-generating-realistic-and-diverse","title":"TrajGen: Generating Realistic and Diverse Trajectories with Reactive and Feasible Agent Behaviors for Autonomous Driving","date":"2022-03-31","arxiv_id":"2203.16792","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-tactile-multimodality-for-following","title":"Visual-Tactile Multimodality for Following Deformable Linear Objects Using Reinforcement Learning","date":"2022-03-31","arxiv_id":"2204.00117","repositories_listed":0,"syntology":null},{"url":null,"slug":"factored-adaptation-for-non-stationary","title":"Factored Adaptation for Non-Stationary Reinforcement Learning","date":"2022-03-30","arxiv_id":"2203.16582","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-properties-of-neural","title":"Investigating the Properties of Neural Network Representations in Reinforcement Learning","date":"2022-03-30","arxiv_id":"2203.15955","repositories_listed":0,"syntology":null},{"url":null,"slug":"marginalized-operators-for-off-policy","title":"Marginalized Operators for Off-policy Reinforcement Learning","date":"2022-03-30","arxiv_id":"2203.16177","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-robot-active-mapping-via-neural","title":"Multi-Robot Active Mapping via Neural Bipartite Graph Matching","date":"2022-03-30","arxiv_id":"2203.16319","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-interpretable-deep-reinforcement","title":"Towards Interpretable Deep Reinforcement Learning Models via Inverse Reinforcement Learning","date":"2022-03-30","arxiv_id":"2203.16464","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-evolutionary-terrain-generation","title":"Assessing Evolutionary Terrain Generation Methods for Curriculum Reinforcement Learning","date":"2022-03-29","arxiv_id":"2203.15172","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-data-driven","title":"Deep Reinforcement Learning for Data-Driven Adaptive Scanning in Ptychography","date":"2022-03-29","arxiv_id":"2203.15413","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-asynchronous-cooperation-with","title":"Asynchronous, Option-Based Multi-Agent Policy Gradient: A Conditional Reasoning Approach","date":"2022-03-29","arxiv_id":"2203.15925","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-reinforcement-learning-effect-handlers-and","title":"On Reinforcement Learning, Effect Handlers, and the State Monad","date":"2022-03-29","arxiv_id":"2203.15426","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-aided-platoon","title":"Deep Reinforcement Learning Aided Platoon Control Relying on V2X Information","date":"2022-03-28","arxiv_id":"2203.15781","repositories_listed":0,"syntology":null},{"url":null,"slug":"reptile-a-proactive-real-time-deep","title":"REPTILE: A Proactive Real-Time Deep Reinforcement Learning Self-adaptive Framework","date":"2022-03-28","arxiv_id":"2203.14686","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-airborne-wind-energy-with","title":"Optimizing Airborne Wind Energy with Reinforcement Learning","date":"2022-03-27","arxiv_id":"2203.14271","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-intelligent-reflecting-surface","title":"Collaborative Intelligent Reflecting Surface Networks with Multi-Agent Reinforcement Learning","date":"2022-03-26","arxiv_id":"2203.14152","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-evolution-and-deep-reinforcement","title":"Combining Evolution and Deep Reinforcement Learning for Policy Search: a Survey","date":"2022-03-26","arxiv_id":"2203.14009","repositories_listed":0,"syntology":null},{"url":null,"slug":"computationally-efficient-joint-coordination","title":"Computationally efficient joint coordination of multiple electric vehicle charging points using reinforcement learning","date":"2022-03-26","arxiv_id":"2203.14078","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-noises-of-multi-agent-environments","title":"Dynamic Noises of Multi-Agent Environments Can Improve Generalization: Agent-based Models meets Reinforcement Learning","date":"2022-03-26","arxiv_id":"2204.14076","repositories_listed":0,"syntology":null},{"url":null,"slug":"dealing-with-sparse-rewards-using-graph","title":"Dealing with Sparse Rewards Using Graph Neural Networks","date":"2022-03-25","arxiv_id":"2203.13424","repositories_listed":0,"syntology":null},{"url":null,"slug":"quasi-newton-iteration-in-deterministic","title":"Quasi-Newton Iteration in Deterministic Policy Gradient","date":"2022-03-25","arxiv_id":"2203.13854","repositories_listed":0,"syntology":null},{"url":null,"slug":"bellman-residual-orthogonalization-for","title":"Bellman Residual Orthogonalization for Offline Reinforcement Learning","date":"2022-03-24","arxiv_id":"2203.12786","repositories_listed":0,"syntology":null},{"url":null,"slug":"horizon-free-reinforcement-learning-in","title":"Horizon-Free Reinforcement Learning in Polynomial Time: the Power of Stationary Policies","date":"2022-03-24","arxiv_id":"2203.12922","repositories_listed":0,"syntology":null},{"url":null,"slug":"merlin-malware-evasion-with-reinforcement","title":"MERLIN -- Malware Evasion with Reinforcement LearnINg","date":"2022-03-24","arxiv_id":"2203.12980","repositories_listed":0,"syntology":null},{"url":null,"slug":"remember-and-forget-experience-replay-for","title":"Remember and Forget Experience Replay for Multi-Agent Reinforcement Learning","date":"2022-03-24","arxiv_id":"2203.13319","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-skills-through-multiple-adversarial","title":"Advanced Skills through Multiple Adversarial Motion Priors in Reinforcement Learning","date":"2022-03-23","arxiv_id":"2203.14912","repositories_listed":0,"syntology":null},{"url":null,"slug":"novgrid-a-flexible-grid-world-for-evaluating","title":"NovGrid: A Flexible Grid World for Evaluating Agent Response to Novelty","date":"2022-03-23","arxiv_id":"2203.12117","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-allocation-optimization-using","title":"The state-of-the-art review on resource allocation problem using artificial intelligence methods on various computing paradigms","date":"2022-03-23","arxiv_id":"2203.12315","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-primer-on-maximum-causal-entropy-inverse","title":"A Primer on Maximum Causal Entropy Inverse Reinforcement Learning","date":"2022-03-22","arxiv_id":"2203.11409","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainability-in-reinforcement-learning","title":"Explainability in reinforcement learning: perspective and position","date":"2022-03-22","arxiv_id":"2203.11547","repositories_listed":0,"syntology":null},{"url":null,"slug":"linear-convergence-of-a-policy-gradient","title":"Linear convergence of a policy gradient method for some finite horizon continuous time control problems","date":"2022-03-22","arxiv_id":"2203.11758","repositories_listed":0,"syntology":null},{"url":null,"slug":"review-of-metrics-to-measure-the-stability","title":"Review of Metrics to Measure the Stability, Robustness and Resilience of Reinforcement Learning","date":"2022-03-22","arxiv_id":"2203.12048","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-deep-reinforcement-learning","title":"Scalable Deep Reinforcement Learning Algorithms for Mean Field Games","date":"2022-03-22","arxiv_id":"2203.11973","repositories_listed":0,"syntology":null},{"url":null,"slug":"x-men-guaranteed-xor-maximum-entropy","title":"X-MEN: Guaranteed XOR-Maximum Entropy Constrained Inverse Reinforcement Learning","date":"2022-03-22","arxiv_id":"2203.11842","repositories_listed":0,"syntology":null},{"url":null,"slug":"lean-evolutionary-reinforcement-learning-by","title":"Multitask Neuroevolution for Reinforcement Learning with Long and Short Episodes","date":"2022-03-21","arxiv_id":"2203.10844","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-trajectories-for-highway-driving","title":"Optimizing Trajectories for Highway Driving with Offline Reinforcement Learning","date":"2022-03-21","arxiv_id":"2203.10949","repositories_listed":0,"syntology":null},{"url":null,"slug":"entailment-relation-aware-paraphrase","title":"Entailment Relation Aware Paraphrase Generation","date":"2022-03-20","arxiv_id":"2203.10483","repositories_listed":0,"syntology":null},{"url":null,"slug":"explicit-user-manipulation-in-reinforcement","title":"Explicit User Manipulation in Reinforcement Learning Based Recommender Systems","date":"2022-03-20","arxiv_id":"2203.10629","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-of","title":"Hierarchical Reinforcement Learning of Locomotion Policies in Response to Approaching Objects: A Preliminary Study","date":"2022-03-20","arxiv_id":"2203.10616","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-multi-agent-reinforcement-2","title":"Model-based Multi-agent Reinforcement Learning: Recent Progress and Prospects","date":"2022-03-20","arxiv_id":"2203.10603","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-reward-function-in","title":"Reinforcement learning reward function in unmanned aerial vehicle control tasks","date":"2022-03-20","arxiv_id":"2203.10519","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-quantum-policy-gradients-with-an","title":"Policy Gradients using Variational Quantum Circuits","date":"2022-03-20","arxiv_id":"2203.10591","repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-on-asymmetric-a-stable","title":"Thompson Sampling on Asymmetric $α$-Stable Bandits","date":"2022-03-19","arxiv_id":"2203.10214","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-guided-graph","title":"Deep reinforcement learning guided graph neural networks for brain network analysis","date":"2022-03-18","arxiv_id":"2203.10093","repositories_listed":0,"syntology":null},{"url":null,"slug":"infinite-horizon-reach-avoid-zero-sum-games","title":"Infinite-Horizon Reach-Avoid Zero-Sum Games via Deep Reinforcement Learning","date":"2022-03-18","arxiv_id":"2203.10142","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-preserving-reinforcement-learning","title":"Privacy-Preserving Reinforcement Learning Beyond Expectation","date":"2022-03-18","arxiv_id":"2203.10165","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-bayesian-games-for-multi-agent","title":"Risk-Sensitive Bayesian Games for Multi-Agent Reinforcement Learning under Policy Uncertainty","date":"2022-03-18","arxiv_id":"2203.10045","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-adaptive-2","title":"Meta-Reinforcement Learning for the Tuning of PI Controllers: An Offline Approach","date":"2022-03-17","arxiv_id":"2203.09661","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-instance-optimal-pac-reinforcement","title":"Near Instance-Optimal PAC Reinforcement Learning for Deterministic MDPs","date":"2022-03-17","arxiv_id":"2203.09251","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategic-maneuver-and-disruption-with","title":"Strategic Maneuver and Disruption with Reinforcement Learning Approaches for Multi-Agent Coordination","date":"2022-03-17","arxiv_id":"2203.09565","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-frost-hollow-experiments-pavlovian","title":"The Frost Hollow Experiments: Pavlovian Signalling as a Path to Coordination and Communication Between Agents","date":"2022-03-17","arxiv_id":"2203.09498","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-based-caching","title":"A Deep Reinforcement Learning-Based Caching Strategy for IoT Networks with Transient Data","date":"2022-03-16","arxiv_id":"2203.12674","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-multi-agent-reinforcement","title":"A Survey of Multi-Agent Deep Reinforcement Learning with Communication","date":"2022-03-16","arxiv_id":"2203.08975","repositories_listed":0,"syntology":null},{"url":null,"slug":"backpropagation-through-time-and-space","title":"Backpropagation through Time and Space: Learning Numerical Methods with Multi-Agent Reinforcement Learning","date":"2022-03-16","arxiv_id":"2203.08937","repositories_listed":0,"syntology":null},{"url":null,"slug":"lazy-mdps-towards-interpretable-reinforcement","title":"Lazy-MDPs: Towards Interpretable Reinforcement Learning by Learning When to Act","date":"2022-03-16","arxiv_id":"2203.08542","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-differentiable-approach-to-combinatorial","title":"A Differentiable Approach to Combinatorial Optimization using Dataless Neural Networks","date":"2022-03-15","arxiv_id":"2203.08209","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-introduction-to-multi-agent-reinforcement","title":"An Introduction to Multi-Agent Reinforcement Learning and Review of its Application to Autonomous Mobility","date":"2022-03-15","arxiv_id":"2203.07676","repositories_listed":0,"syntology":null},{"url":null,"slug":"bi-manual-manipulation-and-attachment-via-sim","title":"Bi-Manual Manipulation and Attachment via Sim-to-Real Reinforcement Learning","date":"2022-03-15","arxiv_id":"2203.08277","repositories_listed":0,"syntology":null},{"url":null,"slug":"blocks-assemble-learning-to-assemble-with","title":"Blocks Assemble! Learning to Assemble with Large-Scale Structured Reinforcement Learning","date":"2022-03-15","arxiv_id":"2203.13733","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-view-dreaming-multi-view-world-model","title":"Multi-View Dreaming: Multi-View World Model with Contrastive Learning","date":"2022-03-15","arxiv_id":"2203.11024","repositories_listed":0,"syntology":null},{"url":null,"slug":"calibration-of-derivative-pricing-models-a","title":"Calibration of Derivative Pricing Models: a Multi-Agent Reinforcement Learning Perspective","date":"2022-03-14","arxiv_id":"2203.06865","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-model-based-multi-agent","title":"Efficient Model-based Multi-agent Reinforcement Learning via Optimistic Equilibrium Computation","date":"2022-03-14","arxiv_id":"2203.07322","repositories_listed":0,"syntology":null},{"url":null,"slug":"frl-fi-transient-fault-analysis-for-federated","title":"FRL-FI: Transient Fault Analysis for Federated Reinforcement Learning-Based Navigation Systems","date":"2022-03-14","arxiv_id":"2203.07276","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-optimal-control-of","title":"Reinforcement Learning for Optimal Control of a District Cooling Energy Plant","date":"2022-03-14","arxiv_id":"2203.07500","repositories_listed":0,"syntology":null},{"url":null,"slug":"switch-trajectory-transformer-with","title":"Switch Trajectory Transformer with Distributional Value Approximation for Multi-Task Reinforcement Learning","date":"2022-03-14","arxiv_id":"2203.07413","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-multi-agent-pickup-and-delivery-problem","title":"The Multi-Agent Pickup and Delivery Problem: MAPF, MARL and Its Warehouse Applications","date":"2022-03-14","arxiv_id":"2203.07092","repositories_listed":0,"syntology":null},{"url":null,"slug":"dara-dynamics-aware-reward-augmentation-in-1","title":"DARA: Dynamics-Aware Reward Augmentation in Offline Reinforcement Learning","date":"2022-03-13","arxiv_id":"2203.06662","repositories_listed":0,"syntology":null},{"url":null,"slug":"concentration-network-for-reinforcement","title":"Concentration Network for Reinforcement Learning of Large-Scale Multi-Agent Systems","date":"2022-03-12","arxiv_id":"2203.06416","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-phase-encode-selection-for-slice","title":"Active Phase-Encode Selection for Slice-Specific Fast MR Scanning Using a Transformer-Based Deep Reinforcement Learning Framework","date":"2022-03-11","arxiv_id":"2203.05756","repositories_listed":0,"syntology":null}],"record_sha256":"0f4cacf7b3178178a98cfd5cdb67fce7499920bb6a5d0273ba05ff06b7749eff","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}