{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/92","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":92,"pages_in_order":152,"rows_per_page":100,"rows":[9101,9200],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/91","next":"/task/reinforcement-learning-1/papers/93","papers":[{"url":null,"slug":"state-of-the-art-of-reinforcement-learning","title":"State of the Art of Reinforcement Learning","date":"2022-01-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"summarising-and-comparing-agent-dynamics-with","title":"Summarising and Comparing Agent Dynamics with Contrastive Spatiotemporal Abstraction","date":"2022-01-17","arxiv_id":"2201.07749","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-deep-observation-a-systematic-survey","title":"Towards deep observation: A systematic survey on artificial intelligence techniques to monitor fetus via Ultrasound Images","date":"2022-01-17","arxiv_id":"2201.07935","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-family-of-cognitively-realistic-parsing","title":"A Family of Cognitively Realistic Parsing Environments for Deep Reinforcement Learning","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"conqrr-conversational-query-rewriting-for-1","title":"CONQRR: Conversational Query Rewriting for Retrieval with Reinforcement Learning","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"inherently-explainable-reinforcement-learning-1","title":"Inherently Explainable Reinforcement Learning in Natural Language","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"jointly-reinforced-user-simulator-and-task","title":"Jointly Reinforced User Simulator and Task-oriented Dialog System with Simplified Generative Architecture","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-atypical-behavior-temporary","title":"Learning from Atypical Behavior: Temporary Interest Aware Recommendation Based on Reinforcement Learning","date":"2022-01-16","arxiv_id":"2201.05970","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-tail-classification-for-distinctive","title":"Long-Tail Classification for Distinctive Image Captioning: A Simple yet Effective Remedy for Side Effects of Reinforcement Learning","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"must-a-framework-for-training-task-oriented","title":"MUST: A Framework for Training Task-oriented Dialogue Systems with Multiple User SimulaTors","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"parsing-natural-language-into-propositional","title":"Parsing Natural Language into Propositional and First-Order Logic with Dual Reinforcement Learning","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-large-action","title":"Reinforcement Learning with Large Action Spaces for Neural Machine Translation","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-the-roles-of-text-in-text-games","title":"Revisiting the Roles of “Text” in Text Games","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-reinforcement-adaptation-for","title":"Unsupervised Reinforcement Adaptation for Class-Imbalanced TextClassification","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"block-policy-mirror-descent","title":"Block Policy Mirror Descent","date":"2022-01-15","arxiv_id":"2201.05756","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-shared","title":"Deep Reinforcement Learning for Shared Autonomous Vehicles (SAV) Fleet Management","date":"2022-01-15","arxiv_id":"2201.05720","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-and-effective-reinforcement","title":"Interpretable and Effective Reinforcement Learning for Attacking against Graph-based Rumor Detection","date":"2022-01-15","arxiv_id":"2201.05819","repositories_listed":0,"syntology":null},{"url":null,"slug":"profitable-strategy-design-by-using-deep","title":"Profitable Strategy Design by Using Deep Reinforcement Learning for Trades on Cryptocurrency Markets","date":"2022-01-15","arxiv_id":"2201.05906","repositories_listed":0,"syntology":null},{"url":null,"slug":"recursive-least-squares-advantage-actor","title":"Recursive Least Squares Advantage Actor-Critic Algorithms","date":"2022-01-15","arxiv_id":"2201.05918","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-air-combat","title":"Reinforcement Learning based Air Combat Maneuver Generation","date":"2022-01-14","arxiv_id":"2201.05528","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-time-varying","title":"Demystifying Reinforcement Learning in Time-Varying Systems","date":"2022-01-14","arxiv_id":"2201.05560","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-to-solve-np-hard","title":"Reinforcement Learning to Solve NP-hard Problems: an Application to the CVRP","date":"2022-01-14","arxiv_id":"2201.05393","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-reinforcement-learning-an-overview","title":"Automated Reinforcement Learning: An Overview","date":"2022-01-13","arxiv_id":"2201.05000","repositories_listed":0,"syntology":null},{"url":null,"slug":"criticality-based-varying-step-number","title":"Criticality-Based Varying Step-Number Algorithm for Reinforcement Learning","date":"2022-01-13","arxiv_id":"2201.05034","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-mutation-and-crossover-in-genetic","title":"Direct Mutation and Crossover in Genetic Algorithms Applied to Reinforcement Learning Tasks","date":"2022-01-13","arxiv_id":"2201.04815","repositories_listed":0,"syntology":null},{"url":null,"slug":"dyna-t-dyna-q-and-upper-confidence-bounds","title":"Dyna-T: Dyna-Q and Upper Confidence Bounds Applied to Trees","date":"2022-01-12","arxiv_id":"2201.04502","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-echelon-supply-chains-with-uncertain","title":"Multi-echelon Supply Chains with Uncertain Seasonal Demands and Lead Times Using Deep Reinforcement Learning","date":"2022-01-12","arxiv_id":"2201.04651","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-recurrent-reinforcement-learning-crypto","title":"The Recurrent Reinforcement Learning Crypto Agent","date":"2022-01-12","arxiv_id":"2201.04699","repositories_listed":0,"syntology":null},{"url":null,"slug":"toddler-guidance-learning-impacts-of-critical","title":"Toddler-Guidance Learning: Impacts of Critical Period on Multimodal AI Agents","date":"2022-01-12","arxiv_id":"2201.04990","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-reinforcement-learning-a-roadmap","title":"Active Reinforcement Learning -- A Roadmap Towards Curious Classifier Systems for Self-Adaptation","date":"2022-01-11","arxiv_id":"2201.03947","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-reinforcement-learning-autorl-a","title":"Automated Reinforcement Learning (AutoRL): A Survey and Open Problems","date":"2022-01-11","arxiv_id":"2201.03916","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-deep-reinforcement-learning","title":"Benchmarking Deep Reinforcement Learning Algorithms for Vision-based Robotics","date":"2022-01-11","arxiv_id":"2201.04224","repositories_listed":0,"syntology":null},{"url":null,"slug":"pavlovian-signalling-with-general-value","title":"Pavlovian Signalling with General Value Functions in Agent-Agent Temporal Decision Making","date":"2022-01-11","arxiv_id":"2201.03709","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-relabelling-for-combined-reinforcement","title":"STIR$^2$: Reward Relabelling for combined Reinforcement and Imitation Learning on sparse-reward tasks","date":"2022-01-11","arxiv_id":"2201.03834","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-independent-capsule-based-agents-for","title":"Task Independent Capsule-Based Agents for Deep Q-Learning","date":"2022-01-11","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-cooperative-multi-agent","title":"Distributed Cooperative Multi-Agent Reinforcement Learning with Directed Coordination Graph","date":"2022-01-10","arxiv_id":"2201.04962","repositories_listed":0,"syntology":null},{"url":null,"slug":"opportunities-of-hybrid-model-based","title":"Opportunities of Hybrid Model-based Reinforcement Learning for Cell Therapy Manufacturing Process Control","date":"2022-01-10","arxiv_id":"2201.03116","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-of-the-art-of-user-simulation","title":"State of the Art of User Simulation approaches for conversational information retrieval","date":"2022-01-10","arxiv_id":"2201.03435","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-is-offline-two-player-zero-sum-markov","title":"When is Offline Two-Player Zero-Sum Markov Game Solvable?","date":"2022-01-10","arxiv_id":"2201.03522","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-reinforcement-learning-approach-1","title":"A Multi-agent Reinforcement Learning Approach for Efficient Client Selection in Federated Learning","date":"2022-01-09","arxiv_id":"2201.02932","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-policy-loss-and-planning","title":"Assessing Policy, Loss and Planning Combinations in Reinforcement Learning using a New Modular Architecture","date":"2022-01-08","arxiv_id":"2201.02874","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-network-optimization-for-reinforcement","title":"Neural Network Optimization for Reinforcement Learning Tasks Using Sparse Computations","date":"2022-01-07","arxiv_id":"2201.02571","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-for-road","title":"Offline Reinforcement Learning for Road Traffic Control","date":"2022-01-07","arxiv_id":"2201.02381","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-reinforcement-learning-and-inverse","title":"Combining Reinforcement Learning and Inverse Reinforcement Learning for Asset Allocation Recommendations","date":"2022-01-06","arxiv_id":"2201.01874","repositories_listed":0,"syntology":null},{"url":null,"slug":"offsetting-unequal-competition-through-rl","title":"Offsetting Unequal Competition through RL-assisted Incentive Schemes","date":"2022-01-05","arxiv_id":"2201.01450","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-1","title":"Deep Reinforcement Learning, a textbook","date":"2022-01-04","arxiv_id":"2201.02135","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-complex-spatial-behaviours-in-abm-an","title":"Learning Complex Spatial Behaviours in ABM: An Experimental Observational Study","date":"2022-01-04","arxiv_id":"2201.01099","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deeper-understanding-of-state-based-critics","title":"A Deeper Understanding of State-Based Critics in Multi-Agent Reinforcement Learning","date":"2022-01-03","arxiv_id":"2201.01221","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-network-for-q-a-in-an","title":"Actor-Critic Network for Q&A in an Adversarial Environment","date":"2022-01-03","arxiv_id":"2201.00455","repositories_listed":0,"syntology":null},{"url":null,"slug":"execute-order-66-targeted-data-poisoning-for","title":"Execute Order 66: Targeted Data Poisoning for Reinforcement Learning","date":"2022-01-03","arxiv_id":"2201.00762","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-general-equilibria-in-many-agent-1","title":"Analyzing Micro-Founded General Equilibrium Models with Many Agents using Deep Reinforcement Learning","date":"2022-01-03","arxiv_id":"2201.01163","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-task","title":"Reinforcement Learning for Task Specifications with Action-Constraints","date":"2022-01-02","arxiv_id":"2201.00286","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-algorithmic-collusion","title":"Robust Algorithmic Collusion","date":"2022-01-02","arxiv_id":"2201.00345","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-surrogate-assisted-controller-for-expensive","title":"A Surrogate-Assisted Controller for Expensive Evolutionary Reinforcement Learning","date":"2022-01-01","arxiv_id":"2201.00129","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-learning-based-stabilization-of","title":"Joint Learning-Based Stabilization of Multiple Unknown Linear Systems","date":"2022-01-01","arxiv_id":"2201.01387","repositories_listed":0,"syntology":null},{"url":null,"slug":"operator-deep-q-learning-zero-shot-reward","title":"Operator Deep Q-Learning: Zero-Shot Reward Transferring in Reinforcement Learning","date":"2022-01-01","arxiv_id":"2201.00236","repositories_listed":0,"syntology":null},{"url":null,"slug":"symmetry-aware-neural-architecture-for-1","title":"Symmetry-Aware Neural Architecture for Embodied Visual Exploration","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-complementarity-guided-reinforcement","title":"Temporal Complementarity-Guided Reinforcement Learning for Image-to-Video Person Re-Identification","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-pareto-efficient-fairness-utility","title":"Toward Pareto Efficient Fairness-Utility Trade-off inRecommendation through Reinforcement Learning","date":"2022-01-01","arxiv_id":"2201.00140","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-rl-across-observation-feature-spaces-1","title":"Transfer RL across Observation Feature Spaces via Model-Based Regularization","date":"2022-01-01","arxiv_id":"2201.00248","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-of-empirical-sample-complexity","title":"Importance of Empirical Sample Complexity Analysis for Offline Reinforcement Learning","date":"2021-12-31","arxiv_id":"2112.15578","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-entropy-regularized-markov-decision","title":"Robust Entropy-regularized Markov Decision Processes","date":"2021-12-31","arxiv_id":"2112.15364","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-shot-pruning-for-offline-reinforcement","title":"Single-Shot Pruning for Offline Reinforcement Learning","date":"2021-12-31","arxiv_id":"2112.15579","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-convex-optimization-for-provably","title":"Stochastic convex optimization for provably efficient apprenticeship learning","date":"2021-12-31","arxiv_id":"2201.00039","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-graph-aware-reinforcement-learning-to","title":"Using Graph-Aware Reinforcement Learning to Identify Winning Strategies in Diplomacy Games (Student Abstract)","date":"2021-12-31","arxiv_id":"2112.15331","repositories_listed":0,"syntology":null},{"url":null,"slug":"constructing-a-good-behavior-basis-for-1","title":"Constructing a Good Behavior Basis for Transfer using Generalized Policy Updates","date":"2021-12-30","arxiv_id":"2112.15025","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-via","title":"Multi-Agent Reinforcement Learning via Adaptive Kalman Temporal Difference and Successor Representation","date":"2021-12-30","arxiv_id":"2112.15156","repositories_listed":0,"syntology":null},{"url":null,"slug":"reversible-upper-confidence-bound-algorithm","title":"Reversible Upper Confidence Bound Algorithm to Generate Diverse Optimized Candidates","date":"2021-12-30","arxiv_id":"2112.14893","repositories_listed":0,"syntology":null},{"url":null,"slug":"stability-preserving-automatic-tuning-of-pid","title":"Stability-Preserving Automatic Tuning of PID Control with Reinforcement Learning","date":"2021-12-30","arxiv_id":"2112.15187","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-theoretic-analysis-of-temporal","title":"Control Theoretic Analysis of Temporal Difference Learning","date":"2021-12-29","arxiv_id":"2112.14417","repositories_listed":0,"syntology":null},{"url":null,"slug":"ddpg-car-following-model-with-real-world","title":"Modified DDPG car-following model with a real-world human driving experience with CARLA simulator","date":"2021-12-29","arxiv_id":"2112.14602","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-performance-bounds-for-primal-dual","title":"Efficient Performance Bounds for Primal-Dual Reinforcement Learning from Demonstrations","date":"2021-12-28","arxiv_id":"2112.14004","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-learning-for-lifelong-visual","title":"Embodied Learning for Lifelong Visual Perception","date":"2021-12-28","arxiv_id":"2112.14084","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-and-risk-management-via","title":"Robustness and risk management via distributional dynamic programming","date":"2021-12-28","arxiv_id":"2112.15430","repositories_listed":0,"syntology":null},{"url":null,"slug":"2112-13937","title":"Multiagent Model-based Credit Assignment for Continuous Control","date":"2021-12-27","arxiv_id":"2112.13937","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-graph-attention-learning-approach-to","title":"A Graph Attention Learning Approach to Antenna Tilt Optimization","date":"2021-12-27","arxiv_id":"2112.14843","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-reinforcement-learning-find-stackelberg","title":"Can Reinforcement Learning Find Stackelberg-Nash Equilibria in General-Sum Markov Games with Myopic Followers?","date":"2021-12-27","arxiv_id":"2112.13521","repositories_listed":0,"syntology":null},{"url":null,"slug":"reldec-reinforcement-learning-based-decoding","title":"RELDEC: Reinforcement Learning-Based Decoding of Moderate Length LDPC Codes","date":"2021-12-27","arxiv_id":"2112.13934","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-chance","title":"Safe Reinforcement Learning with Chance-constrained Model Predictive Control","date":"2021-12-27","arxiv_id":"2112.13941","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-statistical-complexity-of-interactive","title":"The Statistical Complexity of Interactive Decision Making","date":"2021-12-27","arxiv_id":"2112.13487","repositories_listed":0,"syntology":null},{"url":null,"slug":"abstractions-of-general-reinforcement","title":"Abstractions of General Reinforcement Learning","date":"2021-12-26","arxiv_id":"2112.13404","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuro-symbolic-hierarchical-rule-induction","title":"Neuro-Symbolic Hierarchical Rule Induction","date":"2021-12-26","arxiv_id":"2112.13418","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-planning-complexity-of-general","title":"Reducing Planning Complexity of General Reinforcement Learning with Non-Markovian Abstractions","date":"2021-12-26","arxiv_id":"2112.13386","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-interpretable-reinforcement","title":"A Survey on Interpretable Reinforcement Learning","date":"2021-12-24","arxiv_id":"2112.13112","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-channel-access-via-meta-reinforcement","title":"Dynamic Channel Access via Meta-Reinforcement Learning","date":"2021-12-24","arxiv_id":"2201.09075","repositories_listed":0,"syntology":null},{"url":null,"slug":"rediscovering-affordance-a-reinforcement","title":"Rediscovering Affordance: A Reinforcement Learning Perspective","date":"2021-12-24","arxiv_id":"2112.12886","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-efficiency-of-off-policy","title":"Improving the Efficiency of Off-Policy Reinforcement Learning by Accounting for Past Decisions","date":"2021-12-23","arxiv_id":"2112.12281","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-advantage-networks-for-cooperative","title":"Local Advantage Networks for Cooperative Multi-Agent Reinforcement Learning","date":"2021-12-23","arxiv_id":"2112.12458","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-missing-velocity-information-in","title":"Missing Velocity in Dynamic Obstacle Avoidance based on Deep Reinforcement Learning","date":"2021-12-23","arxiv_id":"2112.12465","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-optimal-power","title":"Deep Reinforcement Learning for Optimal Power Flow with Renewables Using Graph Information","date":"2021-12-22","arxiv_id":"2112.11461","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-augmented-deep-reinforcement-learning","title":"Graph augmented Deep Reinforcement Learning in the GameRLand3D environment","date":"2021-12-22","arxiv_id":"2112.11731","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-scalable-deep-reinforcement-learning-model","title":"A Scalable Deep Reinforcement Learning Model for Online Scheduling Coflows of Multi-Stage Jobs for High Performance Computing","date":"2021-12-21","arxiv_id":"2112.11055","repositories_listed":0,"syntology":null},{"url":null,"slug":"aerial-base-station-positioning-and-power","title":"Aerial Base Station Positioning and Power Control for Securing Communications: A Deep Q-Network Approach","date":"2021-12-21","arxiv_id":"2112.11090","repositories_listed":0,"syntology":null},{"url":null,"slug":"district-cooling-system-control-for-providing","title":"District Cooling System Control for Providing Operating Reserve based on Safe Deep Reinforcement Learning","date":"2021-12-21","arxiv_id":"2112.10949","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-androids-dream-of-electric-fences-safety-1","title":"Do Androids Dream of Electric Fences? Safety-Aware Reinforcement Learning with Latent Shielding","date":"2021-12-21","arxiv_id":"2112.11490","repositories_listed":0,"syntology":null},{"url":null,"slug":"nearly-optimal-policy-optimization-with","title":"Nearly Optimal Policy Optimization with Stable at Any Time Guarantee","date":"2021-12-21","arxiv_id":"2112.10935","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-sequential-batch","title":"Reinforcement Learning based Sequential Batch-sampling for Bayesian Optimal Experimental Design","date":"2021-12-21","arxiv_id":"2112.10944","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-model-for","title":"A deep reinforcement learning model for predictive maintenance planning of road assets: Integrating LCA and LCCA","date":"2021-12-20","arxiv_id":"2112.12589","repositories_listed":0,"syntology":null},{"url":null,"slug":"agpnet-autonomous-grading-policy-network","title":"AGPNet -- Autonomous Grading Policy Network","date":"2021-12-20","arxiv_id":"2112.10877","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-preference-based-reinforcement","title":"Interpretable Preference-based Reinforcement Learning with Tree-Structured Reward Functions","date":"2021-12-20","arxiv_id":"2112.11230","repositories_listed":0,"syntology":null}],"record_sha256":"14af8a41f11bb391fac26de6a9cf6b706235d23ea6436709e735af0cc70ba9b7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}