{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/74","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":74,"pages_in_order":132,"rows_per_page":100,"rows":[7301,7400],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/73","next":"/task/reinforcement-learning/papers/75","papers":[{"url":null,"slug":"galois-boosting-deep-reinforcement-learning","title":"GALOIS: Boosting Deep Reinforcement Learning via Generalizable Logic Synthesis","date":"2022-05-27","arxiv_id":"2205.13728","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-markovian-policies-occupancy-measures","title":"Non-Markovian policies occupancy measures","date":"2022-05-27","arxiv_id":"2205.13950","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-beat-multi-agent-reinforcement-learning","title":"Off-Beat Multi-Agent Reinforcement Learning","date":"2022-05-27","arxiv_id":"2205.13718","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-fair-federated-learning-framework-with","title":"A Fair Federated Learning Framework With Reinforcement Learning","date":"2022-05-26","arxiv_id":"2205.13415","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-for-short","title":"Constrained Reinforcement Learning for Short Video Recommendation","date":"2022-05-26","arxiv_id":"2205.13248","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-guided-hierarchical-reward-mechanism","title":"Physics-Guided Hierarchical Reward Mechanism for Learning-Based Robotic Grasping","date":"2022-05-26","arxiv_id":"2205.13561","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporl-temporal-priors-for-exploration-in-1","title":"SFP: State-free Priors for Exploration in Off-Policy Reinforcement Learning","date":"2022-05-26","arxiv_id":"2205.13528","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experimental-comparison-between-temporal","title":"An Experimental Comparison Between Temporal Difference and Residual Gradient with Neural Network Approximation","date":"2022-05-25","arxiv_id":"2205.12770","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-inference-and-transfer-of-compositional","title":"Fast Inference and Transfer of Compositional Task Structures for Few-shot Task Generalization","date":"2022-05-25","arxiv_id":"2205.12648","repositories_listed":0,"syntology":null},{"url":null,"slug":"maviper-learning-decision-tree-policies-for","title":"MAVIPER: Learning Decision Tree Policies for Interpretable Multi-Agent Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.12449","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-goal-oriented-reinforcement","title":"Near-Optimal Goal-Oriented Reinforcement Learning in Non-Stationary Environments","date":"2022-05-25","arxiv_id":"2205.13044","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-on-graphs-for","title":"Robust Reinforcement Learning on Graphs for Logistics optimization","date":"2022-05-25","arxiv_id":"2205.12888","repositories_listed":0,"syntology":null},{"url":null,"slug":"trust-based-consensus-in-multi-agent","title":"Trust-based Consensus in Multi-Agent Reinforcement Learning Systems","date":"2022-05-25","arxiv_id":"2205.12880","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-hamilton-jacobi-bellman","title":"Distributional Hamilton-Jacobi-Bellman Equations for Continuous-Time Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12184","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-convolutional-reinforcement-learning-1","title":"Graph Convolutional Reinforcement Learning for Collaborative Queuing Agents","date":"2022-05-24","arxiv_id":"2205.12009","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-drive-using-sparse-imitation","title":"Learning to Drive Using Sparse Imitation Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12128","repositories_listed":0,"syntology":null},{"url":null,"slug":"penalized-proximal-policy-optimization-for","title":"Penalized Proximal Policy Optimization for Safe Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.11814","repositories_listed":0,"syntology":null},{"url":null,"slug":"computationally-efficient-horizon-free","title":"Computationally Efficient Horizon-Free Reinforcement Learning for Linear Mixture MDPs","date":"2022-05-23","arxiv_id":"2205.11507","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-reinforcement-learning-on-traffic","title":"Cooperative Reinforcement Learning on Traffic Signal Control","date":"2022-05-23","arxiv_id":"2205.11291","repositories_listed":0,"syntology":null},{"url":null,"slug":"logarithmic-regret-bounds-for-continuous-time","title":"Logarithmic regret bounds for continuous-time average-reward Markov decision processes","date":"2022-05-23","arxiv_id":"2205.11168","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-domain-cyberspace-attack-and-defense","title":"Multiple Domain Cyberspace Attack and Defense Game Based on Reward Randomization Reinforcement Learning","date":"2022-05-23","arxiv_id":"2205.10990","repositories_listed":0,"syntology":null},{"url":null,"slug":"polter-policy-trajectory-ensemble","title":"POLTER: Policy Trajectory Ensemble Regularization for Unsupervised Reinforcement Learning","date":"2022-05-23","arxiv_id":"2205.11357","repositories_listed":0,"syntology":null},{"url":null,"slug":"spreading-factor-and-rssi-for-localization-in","title":"Spreading Factor assisted LoRa Localization with Deep Reinforcement Learning","date":"2022-05-23","arxiv_id":"2205.11428","repositories_listed":0,"syntology":null},{"url":null,"slug":"power-and-accountability-in-reinforcement","title":"Power and accountability in reinforcement learning applications to environmental policy","date":"2022-05-22","arxiv_id":"2205.10911","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-pedestrian-attribute-recognition","title":"Reinforced Pedestrian Attribute Recognition with Group Optimization Reward","date":"2022-05-21","arxiv_id":"2205.14042","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-joint-attacks-on-legged-robots","title":"Adversarial joint attacks on legged robots","date":"2022-05-20","arxiv_id":"2205.10098","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-on-fair-reinforcement-learning-theory","title":"Survey on Fair Reinforcement Learning: Theory and Practice","date":"2022-05-20","arxiv_id":"2205.10032","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-valuation-for-offline-reinforcement","title":"Data Valuation for Offline Reinforcement Learning","date":"2022-05-19","arxiv_id":"2205.09550","repositories_listed":0,"syntology":null},{"url":null,"slug":"il-flow-imitation-learning-from-observation","title":"IL-flOw: Imitation Learning from Observation using Normalizing Flows","date":"2022-05-19","arxiv_id":"2205.09251","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-bandit-architecture-based-on-laser","title":"Parallel bandit architecture based on laser chaos for reinforcement learning","date":"2022-05-19","arxiv_id":"2205.09543","repositories_listed":0,"syntology":null},{"url":null,"slug":"routing-and-placement-of-macros-using-deep","title":"Routing and Placement of Macros using Deep Reinforcement Learning","date":"2022-05-19","arxiv_id":"2205.09289","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-adversarial-attack-in-multi-agent","title":"Sparse Adversarial Attack in Multi-agent Reinforcement Learning","date":"2022-05-19","arxiv_id":"2205.09362","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-explanations-from-deep","title":"Generating Explanations from Deep Reinforcement Learning Using Episodic Memory","date":"2022-05-18","arxiv_id":"2205.08926","repositories_listed":0,"syntology":null},{"url":null,"slug":"market-making-via-reinforcement-learning-in","title":"Market Making via Reinforcement Learning in China Commodity Market","date":"2022-05-18","arxiv_id":"2205.08936","repositories_listed":0,"syntology":null},{"url":null,"slug":"slowly-changing-adversarial-bandit-algorithms","title":"Slowly Changing Adversarial Bandit Algorithms are Efficient for Discounted MDPs","date":"2022-05-18","arxiv_id":"2205.09056","repositories_listed":0,"syntology":null},{"url":null,"slug":"world-value-functions-knowledge","title":"World Value Functions: Knowledge Representation for Multitask Reinforcement Learning","date":"2022-05-18","arxiv_id":"2205.08827","repositories_listed":0,"syntology":null},{"url":null,"slug":"moral-reinforcement-learning-using-actual","title":"Moral reinforcement learning using actual causation","date":"2022-05-17","arxiv_id":"2205.08192","repositories_listed":0,"syntology":null},{"url":null,"slug":"multibit-tries-packet-classification-with","title":"Multibit Tries Packet Classification with Deep Reinforcement Learning","date":"2022-05-17","arxiv_id":"2205.08606","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-blind-ai-in","title":"A Deep Reinforcement Learning Blind AI in DareFightingICE","date":"2022-05-16","arxiv_id":"2205.07444","repositories_listed":0,"syntology":null},{"url":null,"slug":"attacking-and-defending-deep-reinforcement","title":"Attacking and Defending Deep Reinforcement Learning Policies","date":"2022-05-16","arxiv_id":"2205.07626","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-apprenticeship-learning-for-playing","title":"Deep Apprenticeship Learning for Playing Games","date":"2022-05-16","arxiv_id":"2205.07959","repositories_listed":0,"syntology":null},{"url":null,"slug":"kgrgrl-a-user-s-permission-reasoning-method","title":"KGRGRL: A User's Permission Reasoning Method Based on Knowledge Graph Reward Guidance Reinforcement Learning","date":"2022-05-16","arxiv_id":"2205.07502","repositories_listed":0,"syntology":null},{"url":null,"slug":"many-field-packet-classification-with","title":"Many Field Packet Classification with Decomposition and Reinforcement Learning","date":"2022-05-16","arxiv_id":"2205.07973","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-munchausen-reinforcement-learning","title":"$q$-Munchausen Reinforcement Learning","date":"2022-05-16","arxiv_id":"2205.07467","repositories_listed":0,"syntology":null},{"url":null,"slug":"qualitative-differences-between-evolutionary","title":"Qualitative Differences Between Evolutionary Strategies and Reinforcement Learning Methods for Control of Autonomous Agents","date":"2022-05-16","arxiv_id":"2205.07592","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-reinforcement-learning-based-logic","title":"Rethinking Reinforcement Learning based Logic Synthesis","date":"2022-05-16","arxiv_id":"2205.07614","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-on-sky-adaptive-optics-control-using","title":"Towards on-sky adaptive optics control using reinforcement learning","date":"2022-05-16","arxiv_id":"2205.07554","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-method-for-robust","title":"Policy Gradient Method For Robust Reinforcement Learning","date":"2022-05-15","arxiv_id":"2205.07344","repositories_listed":0,"syntology":null},{"url":null,"slug":"romfac-a-robust-mean-field-actor-critic","title":"RoMFAC: A robust mean-field actor-critic reinforcement learning against adversarial perturbations on states","date":"2022-05-15","arxiv_id":"2205.07229","repositories_listed":0,"syntology":null},{"url":"/paper/cliff-diving-exploring-reward-surfaces-in","slug":"cliff-diving-exploring-reward-surfaces-in","title":"Cliff Diving: Exploring Reward Surfaces in Reinforcement Learning Environments","date":"2022-05-14","arxiv_id":"2205.07015","repositories_listed":0,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cliff-diving-exploring-reward-surfaces-in#ran","syntology_url":"https://syntology.ai/paper/2205.07015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.07015"}},"official":null}},{"url":null,"slug":"prefixrl-optimization-of-parallel-prefix","title":"PrefixRL: Optimization of Parallel Prefix Circuits using Deep Reinforcement Learning","date":"2022-05-14","arxiv_id":"2205.07000","repositories_listed":0,"syntology":null},{"url":null,"slug":"qhd-a-brain-inspired-hyperdimensional","title":"Efficient Off-Policy Reinforcement Learning via Brain-Inspired Computing","date":"2022-05-14","arxiv_id":"2205.06978","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-bartering-behaviour-in-multi-agent","title":"Emergent Bartering Behaviour in Multi-Agent Reinforcement Learning","date":"2022-05-13","arxiv_id":"2205.06760","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-safe-reinforcement-learning-a","title":"Provably Safe Reinforcement Learning: Conceptual Analysis, Survey, and Benchmarking","date":"2022-05-13","arxiv_id":"2205.06750","repositories_listed":0,"syntology":null},{"url":null,"slug":"contingency-constrained-economic-dispatch","title":"Contingency-constrained economic dispatch with safe reinforcement learning","date":"2022-05-12","arxiv_id":"2205.06212","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-chaotic-itinerancy-in-laser","title":"Controlling chaotic itinerancy in laser dynamics for reinforcement learning","date":"2022-05-12","arxiv_id":"2205.05987","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-and-instance-joint-selection-a","title":"Feature and Instance Joint Selection: A Reinforcement Learning Perspective","date":"2022-05-12","arxiv_id":"2205.07867","repositories_listed":0,"syntology":null},{"url":null,"slug":"delayed-reinforcement-learning-by-imitation","title":"Delayed Reinforcement Learning by Imitation","date":"2022-05-11","arxiv_id":"2205.05569","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-cooperative-policies-for-multi-1","title":"Developing cooperative policies for multi-stage reinforcement learning tasks","date":"2022-05-11","arxiv_id":"2205.05230","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-distributed-framework-for","title":"Efficient Distributed Framework for Collaborative Multi-Agent Reinforcement Learning","date":"2022-05-11","arxiv_id":"2205.05248","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-guide-multiple-heterogeneous","title":"Learning to Guide Multiple Heterogeneous Actors from a Single Human Demonstration via Automatic Curriculum Learning in StarCraft II","date":"2022-05-11","arxiv_id":"2205.05784","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerated-reinforcement-learning-for-1","title":"Accelerated Reinforcement Learning for Temporal Logic Control Objectives","date":"2022-05-09","arxiv_id":"2205.04424","repositories_listed":0,"syntology":null},{"url":null,"slug":"search-based-testing-of-reinforcement","title":"Search-Based Testing of Reinforcement Learning","date":"2022-05-07","arxiv_id":"2205.04887","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-oriented-next-best-activity","title":"Goal-Oriented Next Best Activity Recommendation using Reinforcement Learning","date":"2022-05-06","arxiv_id":"2205.03219","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approach-to-estimation","title":"Reinforcement Learning Approach to Estimation in Linear Systems","date":"2022-05-06","arxiv_id":"2205.03504","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-temporal-pattern-backdoor-attack-to-deep","title":"A Temporal-Pattern Backdoor Attack to Deep Reinforcement Learning","date":"2022-05-05","arxiv_id":"2205.02589","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-in","title":"Multi-Agent Deep Reinforcement Learning in Vehicular OCC","date":"2022-05-05","arxiv_id":"2205.02672","repositories_listed":0,"syntology":null},{"url":null,"slug":"rapid-locomotion-via-reinforcement-learning","title":"Rapid Locomotion via Reinforcement Learning","date":"2022-05-05","arxiv_id":"2205.02824","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-subgoal-robot-navigation-in-crowds-with","title":"Multi-subgoal Robot Navigation in Crowds with History Information and Interactions","date":"2022-05-04","arxiv_id":"2205.02003","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-representation-learning-for-goal","title":"State Representation Learning for Goal-Conditioned Reinforcement Learning","date":"2022-05-04","arxiv_id":"2205.01965","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-attack-over-the-deep-reinforcement","title":"Deep-Attack over the Deep Reinforcement Learning","date":"2022-05-02","arxiv_id":"2205.00807","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-in-deep-reinforcement-learning-a-1","title":"Exploration in Deep Reinforcement Learning: A Survey","date":"2022-05-02","arxiv_id":"2205.00824","repositories_listed":0,"syntology":null},{"url":null,"slug":"processing-network-controls-via-deep","title":"Processing Network Controls via Deep Reinforcement Learning","date":"2022-05-01","arxiv_id":"2205.02119","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-reinforcement-learning-for","title":"Unsupervised Reinforcement Learning for Transferable Manipulation Skill Discovery","date":"2022-04-29","arxiv_id":"2204.13906","repositories_listed":0,"syntology":null},{"url":null,"slug":"riscless-a-reinforcement-learning-strategy-to","title":"RISCLESS: A Reinforcement Learning Strategy to Exploit Unused Cloud Resources","date":"2022-04-28","arxiv_id":"2205.08350","repositories_listed":0,"syntology":null},{"url":null,"slug":"bisimulation-makes-analogies-in-goal","title":"Bisimulation Makes Analogies in Goal-Conditioned Reinforcement Learning","date":"2022-04-27","arxiv_id":"2204.13060","repositories_listed":0,"syntology":null},{"url":null,"slug":"relational-abstractions-for-generalized","title":"Relational Abstractions for Generalized Reinforcement Learning on Symbolic Problems","date":"2022-04-27","arxiv_id":"2204.12665","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-efficient-dynamic-sampling-policy-for","title":"An Efficient Dynamic Sampling Policy For Monte Carlo Tree Search","date":"2022-04-26","arxiv_id":"2204.12043","repositories_listed":0,"syntology":null},{"url":null,"slug":"bats-best-action-trajectory-stitching","title":"BATS: Best Action Trajectory Stitching","date":"2022-04-26","arxiv_id":"2204.12026","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-online-3","title":"Deep Reinforcement Learning for Online Routing of Unmanned Aerial Vehicles with Wireless Power Transfer","date":"2022-04-25","arxiv_id":"2204.11477","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-orienteering","title":"Deep Reinforcement Learning for Orienteering Problems Based on Decomposition","date":"2022-04-25","arxiv_id":"2204.11575","repositories_listed":0,"syntology":null},{"url":null,"slug":"skill-based-meta-reinforcement-learning-1","title":"Skill-based Meta-Reinforcement Learning","date":"2022-04-25","arxiv_id":"2204.11828","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-radio","title":"Deep Reinforcement Learning-based Radio Resource Allocation and Beam Management under Location Uncertainty in 5G mmWave Networks","date":"2022-04-23","arxiv_id":"2204.10984","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-neural-network-based-agent-in-google","title":"Graph Neural Network based Agent in Google Research Football","date":"2022-04-23","arxiv_id":"2204.11142","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-how-to-interact-with-a-complex","title":"Learning how to Interact with a Complex Interface using Hierarchical Reinforcement Learning","date":"2022-04-21","arxiv_id":"2204.10374","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-nitrogen-management-with-deep","title":"Optimizing Nitrogen Management with Deep Reinforcement Learning and Crop Simulations","date":"2022-04-21","arxiv_id":"2204.10394","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-bayesian-approach-to-inverse-1","title":"A Hierarchical Bayesian Approach to Inverse Reinforcement Learning with Symbolic Reward Machines","date":"2022-04-20","arxiv_id":"2204.09772","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-learning-for-distributed-energy","title":"Federated Learning for Distributed Energy-Efficient Resource Allocation","date":"2022-04-20","arxiv_id":"2204.09602","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-intrinsic","title":"Reinforcement Learning with Intrinsic Affinity for Personalized Prosperity Management","date":"2022-04-20","arxiv_id":"2204.09218","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-for-1","title":"Reinforcement Learning from Partial Observation: Linear Function Approximation with Provable Sample Efficiency","date":"2022-04-20","arxiv_id":"2204.09787","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-and-preventing-capacity-loss-in-1","title":"Understanding and Preventing Capacity Loss in Reinforcement Learning","date":"2022-04-20","arxiv_id":"2204.09560","repositories_listed":0,"syntology":null},{"url":null,"slug":"network-topology-optimization-via-deep","title":"Network Topology Optimization via Deep Reinforcement Learning","date":"2022-04-19","arxiv_id":"2204.14133","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-is-partially-observable-reinforcement","title":"When Is Partially Observable Reinforcement Learning Not Scary?","date":"2022-04-19","arxiv_id":"2204.08967","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-tensor-network-contraction-using","title":"Optimizing Tensor Network Contraction Using Reinforcement Learning","date":"2022-04-18","arxiv_id":"2204.09052","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-transfer-role-assignment-across","title":"Learning to Transfer Role Assignment Across Team Sizes","date":"2022-04-17","arxiv_id":"2204.12937","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-charging-power-forecast-of-evcs","title":"Probabilistic Charging Power Forecast of EVCS: Reinforcement Learning Assisted Deep Learning Approach","date":"2022-04-17","arxiv_id":"2204.07905","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-for","title":"Efficient Reinforcement Learning for Unsupervised Controlled Text Generation","date":"2022-04-16","arxiv_id":"2204.07696","repositories_listed":0,"syntology":null},{"url":null,"slug":"cryorl-reinforcement-learning-enables","title":"CryoRL: Reinforcement Learning Enables Efficient Cryo-EM Data Collection","date":"2022-04-15","arxiv_id":"2204.07543","repositories_listed":0,"syntology":null},{"url":null,"slug":"d3pg-dirichlet-ddpg-for-task-partitioning-and","title":"D3PG: Dirichlet DDPG for Task Partitioning and Offloading With Constrained Hybrid Action Space in Mobile-Edge Computing","date":"2022-04-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"methodical-advice-collection-and-reuse-in","title":"Methodical Advice Collection and Reuse in Deep Reinforcement Learning","date":"2022-04-14","arxiv_id":"2204.07254","repositories_listed":0,"syntology":null}],"record_sha256":"5f8fb13301d4fbff7b0bbff2fa3023c82c3529801e9c8366bf779b9487889562","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}