{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/99","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":99,"pages_in_order":152,"rows_per_page":100,"rows":[9801,9900],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/98","next":"/task/reinforcement-learning-1/papers/100","papers":[{"url":null,"slug":"uncertainty-regularized-policy-learning-for","title":"Uncertainty Regularized Policy Learning for Offline Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-and-leveraging","title":"Understanding and Leveraging Overparameterization in Recursive Value Estimation","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-generalization-gap-in","title":"Understanding the Generalization Gap in Visual Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"untangling-braids-with-multi-agent-q-learning","title":"Untangling Braids with Multi-agent Q-Learning","date":"2021-09-29","arxiv_id":"2109.14502","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-refinement-network-vrn","title":"Value Refinement Network (VRN)","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-oracle-guiding-for-reinforcement","title":"Variational oracle guiding for reinforcement learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"wavecorr-deep-reinforcement-learning-with","title":"WaveCorr: Deep Reinforcement Learning with Permutation Invariant Policy Networks for Portfolio Management","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-learning-of-disentangled","title":"Weakly-Supervised Learning of Disentangled and Interpretable Skills for Hierarchical Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"why-so-pessimistic-estimating-uncertainties","title":"Why so pessimistic? Estimating uncertainties for offline RL through ensembles, and why their independence matters.","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-reward-specification-via-grounded","title":"Zero-Shot Reward Specification via Grounded Natural Language","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-first-occupancy-representation-for","title":"A First-Occupancy Representation for Reinforcement Learning","date":"2021-09-28","arxiv_id":"2109.13863","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-informative-path-planning-using-deep","title":"Adaptive Informative Path Planning Using Deep Reinforcement Learning for UAV-based Active Sensing","date":"2021-09-28","arxiv_id":"2109.13570","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-offline-deep-reinforcement-learning-for","title":"An Offline Deep Reinforcement Learning for Maintenance Decision-Making","date":"2021-09-28","arxiv_id":"2109.15050","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-versus-evolution","title":"Deep Reinforcement Learning Versus Evolution Strategies: A Comparative Survey","date":"2021-09-28","arxiv_id":"2110.01411","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-adjustments","title":"Deep Reinforcement Learning with Adjustments","date":"2021-09-28","arxiv_id":"2109.13463","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-more-when-it-needs-in-deep","title":"Exploring More When It Needs in Deep Reinforcement Learning","date":"2021-09-28","arxiv_id":"2109.13477","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-reasoning-flaws-in-planning-based","title":"Identifying Reasoning Flaws in Planning-Based RL Using Tree Explanations","date":"2021-09-28","arxiv_id":"2109.13978","repositories_listed":0,"syntology":null},{"url":null,"slug":"longitudinal-deep-truck-deep-learning-and","title":"Longitudinal Deep Truck: Deep learning and deep reinforcement learning for modeling and control of longitudinal dynamics of heavy duty trucks","date":"2021-09-28","arxiv_id":"2109.14019","repositories_listed":0,"syntology":null},{"url":null,"slug":"making-curiosity-explicit-in-vision-based-rl","title":"Making Curiosity Explicit in Vision-based RL","date":"2021-09-28","arxiv_id":"2109.13588","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-quantitative","title":"Reinforcement Learning for Quantitative Trading","date":"2021-09-28","arxiv_id":"2109.13851","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-based-slice-placement-under-realistic","title":"DRL-based Slice Placement under Realistic Network Load Conditions","date":"2021-09-27","arxiv_id":"2109.12857","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficiently-training-on-policy-actor-critic","title":"Efficiently Training On-Policy Actor-Critic Networks in Robotic Deep Reinforcement Learning with Demonstration-like Sampled Exploration","date":"2021-09-27","arxiv_id":"2109.13005","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-internal-models-toward-metacognitive-ai","title":"From internal models toward metacognitive AI","date":"2021-09-27","arxiv_id":"2109.12798","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-reinforcement-learning-for-optimal","title":"Model-Free Reinforcement Learning for Optimal Control of MarkovDecision Processes Under Signal Temporal Logic Specifications","date":"2021-09-27","arxiv_id":"2109.13377","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-reinforcement-learning-for-pivot","title":"Towards Reinforcement Learning for Pivot-based Neural Machine Translation with Non-autoregressive Transformer","date":"2021-09-27","arxiv_id":"2109.13097","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-wireless-3","title":"Deep Reinforcement Learning for Wireless Scheduling in Distributed Networked Control","date":"2021-09-26","arxiv_id":"2109.12562","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-feasibility-of-learning-finger-gaiting","title":"On the Feasibility of Learning Finger-gaiting In-hand Manipulation with Intrinsic Sensing","date":"2021-09-26","arxiv_id":"2109.12720","repositories_listed":0,"syntology":null},{"url":null,"slug":"l-2-nas-learning-to-optimize-neural","title":"L$^{2}$NAS: Learning to Optimize Neural Architectures via Continuous-Action Reinforcement Learning","date":"2021-09-25","arxiv_id":"2109.12425","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-graph-policy-network-approach-for-volt-var","title":"A Graph Policy Network Approach for Volt-Var Control in Power Distribution Systems","date":"2021-09-24","arxiv_id":"2109.12073","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-sampling-quasi-newton-methods-for-1","title":"Adaptive Sampling Quasi-Newton Methods for Zeroth-Order Stochastic Optimization","date":"2021-09-24","arxiv_id":"2109.12213","repositories_listed":0,"syntology":null},{"url":null,"slug":"combing-policy-evaluation-and-policy","title":"The $f$-Divergence Reinforcement Learning Framework","date":"2021-09-24","arxiv_id":"2109.11867","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-deep-reinforcement-learning-for-5","title":"Combining Contention-Based Spectrum Access and Adaptive Modulation using Deep Reinforcement Learning","date":"2021-09-24","arxiv_id":"2109.11723","repositories_listed":0,"syntology":null},{"url":null,"slug":"go-blend-behavior-and-affect","title":"Go-Blend behavior and affect","date":"2021-09-24","arxiv_id":"2109.13388","repositories_listed":0,"syntology":null},{"url":null,"slug":"learnable-triangulation-for-deep-learning","title":"Learnable Triangulation for Deep Learning-based 3D Reconstruction of Objects of Arbitrary Topology from Single RGB Images","date":"2021-09-24","arxiv_id":"2109.11844","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuroprospecting-with-deeprl-agents","title":"Neuroprospecting with DeepRL agents","date":"2021-09-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"regularization-guarantees-generalization-in","title":"Regularization Guarantees Generalization in Bayesian Reinforcement Learning through Algorithmic Stability","date":"2021-09-24","arxiv_id":"2109.11792","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-deep-reinforcement-learning-2","title":"A Multi-Agent Deep Reinforcement Learning Coordination Framework for Connected and Automated Vehicles at Merging Roadways","date":"2021-09-23","arxiv_id":"2109.11672","repositories_listed":0,"syntology":null},{"url":null,"slug":"dimension-free-rates-for-natural-policy","title":"Dimension-Free Rates for Natural Policy Gradient in Multi-Agent Reinforcement Learning","date":"2021-09-23","arxiv_id":"2109.11692","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchies-of-planning-and-reinforcement","title":"Hierarchies of Planning and Reinforcement Learning for Robot Navigation","date":"2021-09-23","arxiv_id":"2109.11178","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-path-planning-for-long-range","title":"Deep Reinforcement Learning-Based Long-Range Autonomous Valet Parking for Smart Cities","date":"2021-09-23","arxiv_id":"2109.11661","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictionnet-real-time-joint-probabilistic","title":"PredictionNet: Real-Time Joint Probabilistic Traffic Prediction for Planning, Control, and Simulation","date":"2021-09-23","arxiv_id":"2109.11094","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-under-algorithmic","title":"Reinforcement Learning Under Algorithmic Triage","date":"2021-09-23","arxiv_id":"2109.11328","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-reinforcement-learning-for-1","title":"A Survey on Reinforcement Learning for Recommender Systems","date":"2021-09-22","arxiv_id":"2109.10665","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-training-blocks-generalization-in","title":"Adversarial Training Blocks Generalization in Neural Policies","date":"2021-09-22","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-lane-changing-decision-making","title":"Benchmarking Lane-changing Decision-making for Deep Reinforcement Learning","date":"2021-09-22","arxiv_id":"2109.10490","repositories_listed":0,"syntology":null},{"url":null,"slug":"introducing-symmetries-to-black-box-meta","title":"Introducing Symmetries to Black Box Meta Reinforcement Learning","date":"2021-09-22","arxiv_id":"2109.10781","repositories_listed":0,"syntology":null},{"url":null,"slug":"locality-matters-a-scalable-value","title":"Locality Matters: A Scalable Value Decomposition Approach for Cooperative Multi-Agent Reinforcement Learning","date":"2021-09-22","arxiv_id":"2109.10632","repositories_listed":0,"syntology":null},{"url":null,"slug":"mepg-a-minimalist-ensemble-policy-gradient","title":"MEPG: A Minimalist Ensemble Policy Gradient Framework for Deep Reinforcement Learning","date":"2021-09-22","arxiv_id":"2109.10552","repositories_listed":0,"syntology":null},{"url":null,"slug":"return-dispersion-as-an-estimator-of-learning","title":"Return Dispersion as an Estimator of Learning Potential for Prioritized Level Replay","date":"2021-09-22","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-multi-agent-reinforcement-learning-1","title":"Towards Multi-Agent Reinforcement Learning using Quantum Boltzmann Machines","date":"2021-09-22","arxiv_id":"2109.10900","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-unified-framework-for-anomaly","title":"A Distance-based Anomaly Detection Framework for Deep Reinforcement Learning","date":"2021-09-21","arxiv_id":"2109.09889","repositories_listed":0,"syntology":null},{"url":null,"slug":"example-driven-model-based-reinforcement","title":"Example-Driven Model-Based Reinforcement Learning for Solving Long-Horizon Visuomotor Tasks","date":"2021-09-21","arxiv_id":"2109.10312","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-offline-memory-replay-in-biological","title":"Learning offline: memory replay in biological and artificial reinforcement learning","date":"2021-09-21","arxiv_id":"2109.10034","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-to-the-1","title":"A Reinforcement Learning Approach to the Stochastic Cutting Stock Problem","date":"2021-09-20","arxiv_id":"2109.09592","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-text-games-for-reinforcement","title":"A Survey of Text Games for Reinforcement Learning informed by Natural Language","date":"2021-09-20","arxiv_id":"2109.09478","repositories_listed":0,"syntology":null},{"url":null,"slug":"carl-conditional-value-at-risk-adversarial","title":"ACReL: Adversarial Conditional value-at-risk Reinforcement Learning","date":"2021-09-20","arxiv_id":"2109.09470","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-natural-language-generation-from","title":"Learning Natural Language Generation from Scratch","date":"2021-09-20","arxiv_id":"2109.09371","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-finite-horizon","title":"Reinforcement Learning for Finite-Horizon Restless Multi-Armed Multi-Action Bandits","date":"2021-09-20","arxiv_id":"2109.09855","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-approaches-to-building-collaborative-task","title":"Two Approaches to Building Collaborative, Task-Oriented Dialog Agents through Self-Play","date":"2021-09-20","arxiv_id":"2109.09597","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-behavior-regularized-reinforcement","title":"Dual Behavior Regularized Reinforcement Learning","date":"2021-09-19","arxiv_id":"2109.09037","repositories_listed":0,"syntology":null},{"url":null,"slug":"greedy-unmixing-for-q-learning-in-multi-agent","title":"Greedy UnMixing for Q-Learning in Multi-Agent Reinforcement Learning","date":"2021-09-19","arxiv_id":"2109.09034","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifelong-robotic-reinforcement-learning-by","title":"Lifelong Robotic Reinforcement Learning by Retaining Experiences","date":"2021-09-19","arxiv_id":"2109.09180","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularize-don-t-mix-multi-agent","title":"Regularize! Don't Mix: Multi-Agent Reinforcement Learning without Explicit Centralized Structures","date":"2021-09-19","arxiv_id":"2109.09038","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-offline-reinforcement-learning","title":"Accelerating Offline Reinforcement Learning Application in Real-Time Bidding and Recommendation: Potential Use of Simulation","date":"2021-09-17","arxiv_id":"2109.08331","repositories_listed":0,"syntology":null},{"url":null,"slug":"carl-lead-lidar-based-end-to-end-autonomous","title":"Carl-Lead: Lidar-based End-to-End Autonomous Driving with Contrastive Deep Reinforcement Learning","date":"2021-09-17","arxiv_id":"2109.08473","repositories_listed":0,"syntology":null},{"url":null,"slug":"coordinated-random-access-for-industrial-iot","title":"Coordinated Random Access for Industrial IoT With Correlated Traffic By Reinforcement-Learning","date":"2021-09-17","arxiv_id":"2109.08389","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-global-connectivity-maintenance","title":"Decentralized Global Connectivity Maintenance for Multi-Robot Navigation: A Reinforcement Learning Approach","date":"2021-09-17","arxiv_id":"2109.08536","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-2","title":"Deep Reinforcement Learning Based Multidimensional Resource Management for Energy Harvesting Cognitive NOMA Communications","date":"2021-09-17","arxiv_id":"2109.09503","repositories_listed":0,"syntology":null},{"url":null,"slug":"soft-actor-critic-with-integer-actions","title":"Soft Actor-Critic With Integer Actions","date":"2021-09-17","arxiv_id":"2109.08512","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-and-unification-of-three","title":"Comparison and Unification of Three Regularization Methods in Batch Reinforcement Learning","date":"2021-09-16","arxiv_id":"2109.08134","repositories_listed":0,"syntology":null},{"url":null,"slug":"conservative-data-sharing-for-multi-task","title":"Conservative Data Sharing for Multi-Task Offline Reinforcement Learning","date":"2021-09-16","arxiv_id":"2109.08128","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-risk-aware-reinforcement-learning","title":"Enabling risk-aware Reinforcement Learning for medical interventions through uncertainty decomposition","date":"2021-09-16","arxiv_id":"2109.07827","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-peers-transfer-reinforcement","title":"Learning from Peers: Deep Transfer Reinforcement Learning for Joint Radio and Cache Resource Allocation in 5G RAN Slicing","date":"2021-09-16","arxiv_id":"2109.07999","repositories_listed":0,"syntology":null},{"url":null,"slug":"rapid-rl-a-reconfigurable-architecture-with","title":"RAPID-RL: A Reconfigurable Architecture with Preemptive-Exits for Efficient Deep-Reinforcement Learning","date":"2021-09-16","arxiv_id":"2109.08231","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-on-encrypted-data","title":"Reinforcement Learning on Encrypted Data","date":"2021-09-16","arxiv_id":"2109.08236","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-of-a-human-in-the-loop-policy","title":"Convergence of a Human-in-the-Loop Policy-Gradient Algorithm With Eligibility Trace Under Reward, Policy, and Advantage Feedback","date":"2021-09-15","arxiv_id":"2109.07054","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-cycling-of-a-heterogenous-battery","title":"Optimal Cycling of a Heterogenous Battery Bank via Reinforcement Learning","date":"2021-09-15","arxiv_id":"2109.07137","repositories_listed":0,"syntology":null},{"url":null,"slug":"short-quantum-circuits-in-reinforcement","title":"Short Quantum Circuits in Reinforcement Learning Policies for the Vehicle Routing Problem","date":"2021-09-15","arxiv_id":"2109.07498","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-does-the-user-want-information-gain-for","title":"What Does The User Want? Information Gain for Hierarchical Dialogue Policy Optimisation","date":"2021-09-15","arxiv_id":"2109.07129","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-homeostatic-reinforcement-learning","title":"Continuous Homeostatic Reinforcement Learning for Self-Regulated Autonomous Agents","date":"2021-09-14","arxiv_id":"2109.06580","repositories_listed":0,"syntology":null},{"url":null,"slug":"dsdf-an-approach-to-handle-stochastic-agents","title":"DSDF: An approach to handle stochastic agents in collaborative multi-agent reinforcement learning","date":"2021-09-14","arxiv_id":"2109.06609","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-in-deep-reinforcement-learning-a","title":"Exploration in Deep Reinforcement Learning: From Single-Agent to Multiagent Domain","date":"2021-09-14","arxiv_id":"2109.06668","repositories_listed":0,"syntology":null},{"url":null,"slug":"romax-certifiably-robust-deep-multiagent","title":"ROMAX: Certifiably Robust Deep Multiagent Reinforcement Learning via Convex Relaxation","date":"2021-09-14","arxiv_id":"2109.06795","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-practical-adversarial-attack-on-contingency","title":"A Practical Adversarial Attack on Contingency Detection of Smart Energy Systems","date":"2021-09-13","arxiv_id":"2109.06358","repositories_listed":0,"syntology":null},{"url":null,"slug":"achieving-zero-constraint-violation-for","title":"Achieving Zero Constraint Violation for Constrained Reinforcement Learning via Primal-Dual Approach","date":"2021-09-13","arxiv_id":"2109.06332","repositories_listed":0,"syntology":null},{"url":null,"slug":"computation-rate-maximum-for-mobile-terminals","title":"Computation Rate Maximum for Mobile Terminals in UAV-assisted Wireless Powered MEC Networks with Fairness Constraint","date":"2021-09-13","arxiv_id":"2109.05767","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-emptive-learning-to-defer-for-sequential","title":"Learning-to-defer for sequential medical decision-making under uncertainty","date":"2021-09-13","arxiv_id":"2109.06312","repositories_listed":0,"syntology":null},{"url":null,"slug":"radars-memory-efficient-reinforcement","title":"RADARS: Memory Efficient Reinforcement Learning Aided Differentiable Neural Architecture Search","date":"2021-09-13","arxiv_id":"2109.05691","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-load-balanced","title":"Reinforcement Learning for Load-balanced Parallel Particle Tracing","date":"2021-09-13","arxiv_id":"2109.05679","repositories_listed":0,"syntology":null},{"url":null,"slug":"theoretical-guarantees-of-fictitious-discount","title":"Theoretical Guarantees of Fictitious Discount Algorithms for Episodic Reinforcement Learning and Global Convergence of Policy Gradient Methods","date":"2021-09-13","arxiv_id":"2109.06362","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-socially-aware-reinforcement-learning-agent","title":"A Socially Aware Reinforcement Learning Agent for The Single Track Road Problem","date":"2021-09-12","arxiv_id":"2109.05486","repositories_listed":0,"syntology":null},{"url":null,"slug":"concave-utility-reinforcement-learning-with","title":"Concave Utility Reinforcement Learning with Zero-Constraint Violations","date":"2021-09-12","arxiv_id":"2109.05439","repositories_listed":0,"syntology":null},{"url":null,"slug":"emvlight-a-decentralized-reinforcement","title":"EMVLight: A Decentralized Reinforcement Learning Framework for Efficient Passage of Emergency Vehicles","date":"2021-09-12","arxiv_id":"2109.05429","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-ensemble-model-based-reinforcement","title":"Federated Ensemble Model-based Reinforcement Learning in Edge Computing","date":"2021-09-12","arxiv_id":"2109.05549","repositories_listed":0,"syntology":null},{"url":null,"slug":"financial-trading-with-feature-preprocessing","title":"Financial Trading with Feature Preprocessing and Recurrent Reinforcement Learning","date":"2021-09-11","arxiv_id":"2109.05283","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-generation-method-for-learning-a-low","title":"Data Generation Method for Learning a Low-dimensional Safe Region in Safe Reinforcement Learning","date":"2021-09-10","arxiv_id":"2109.05077","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-madrl","title":"Multi-agent deep reinforcement learning (MADRL) meets multi-user MIMO systems","date":"2021-09-10","arxiv_id":"2109.04986","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-a-domestic-battery-and-solar","title":"Optimizing a domestic battery and solar photovoltaic system with deep reinforcement learning","date":"2021-09-10","arxiv_id":"2109.05024","repositories_listed":0,"syntology":null},{"url":null,"slug":"projected-state-action-balancing-weights-for","title":"Projected State-action Balancing Weights for Offline Reinforcement Learning","date":"2021-09-10","arxiv_id":"2109.04640","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-equal-risk","title":"Deep Reinforcement Learning for Equal Risk Pricing and Hedging under Dynamic Expectile Risk Measures","date":"2021-09-09","arxiv_id":"2109.04001","repositories_listed":0,"syntology":null}],"record_sha256":"1a26f8b7473db5307919d5de7507c557c0a44bf73dd53afe672848ed2679c7f3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}