{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/105","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":105,"pages_in_order":152,"rows_per_page":100,"rows":[10401,10500],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/104","next":"/task/reinforcement-learning-1/papers/106","papers":[{"url":null,"slug":"reinforcement-learning-based-safe-decision","title":"Reinforcement Learning Based Safe Decision Making for Highway Autonomous Driving","date":"2021-05-13","arxiv_id":"2105.06517","repositories_listed":0,"syntology":null},{"url":null,"slug":"side-i-infer-the-state-i-want-to-learn","title":"SIDE: State Inference for Partially Observable Cooperative Multi-Agent Reinforcement Learning","date":"2021-05-13","arxiv_id":"2105.06228","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-reinforcement-learning-aided","title":"A Survey on Reinforcement Learning-Aided Caching in Mobile Edge Networks","date":"2021-05-12","arxiv_id":"2105.05564","repositories_listed":0,"syntology":null},{"url":null,"slug":"acting-upon-imagination-when-to-trust","title":"Acting upon Imagination: when to trust imagined trajectories in model based reinforcement learning","date":"2021-05-12","arxiv_id":"2105.05716","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-reinforcement-learning-in-dynamic","title":"Adversarial Reinforcement Learning in Dynamic Channel Access and Power Control","date":"2021-05-12","arxiv_id":"2105.05817","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-performance-analysis-towards","title":"Interpretable performance analysis towards offline reinforcement learning: A dataset perspective","date":"2021-05-12","arxiv_id":"2105.05473","repositories_listed":0,"syntology":null},{"url":null,"slug":"composable-energy-policies-for-reactive","title":"Composable Energy Policies for Reactive Motion Generation and Reinforcement Learning","date":"2021-05-11","arxiv_id":"2105.04962","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-rnns-based-transformers-maddpg","title":"Hierarchical RNNs-Based Transformers MADDPG for Mixed Cooperative-Competitive Environments","date":"2021-05-11","arxiv_id":"2105.04888","repositories_listed":0,"syntology":null},{"url":null,"slug":"return-based-scaling-yet-another","title":"Return-based Scaling: Yet Another Normalisation Trick for Deep RL","date":"2021-05-11","arxiv_id":"2105.05347","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-reinforcement-learning-on-graphs","title":"Zero-Shot Reinforcement Learning on Graphs for Autonomous Exploration Under Uncertainty","date":"2021-05-11","arxiv_id":"2105.04758","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-policy-transfer-in-reinforcement","title":"Adaptive Policy Transfer in Reinforcement Learning","date":"2021-05-10","arxiv_id":"2105.04699","repositories_listed":0,"syntology":null},{"url":null,"slug":"age-of-information-aware-vnf-scheduling-in","title":"Age of Information Aware VNF Scheduling in Industrial IoT Using Deep Reinforcement Learning","date":"2021-05-10","arxiv_id":"2105.04207","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-multichannel-access-via-multi-agent","title":"Dynamic Multichannel Access via Multi-agent Reinforcement Learning: Throughput and Fairness Guarantees","date":"2021-05-10","arxiv_id":"2105.04077","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-self-supervised-data-collection-for","title":"Efficient Self-Supervised Data Collection for Offline Robot Learning","date":"2021-05-10","arxiv_id":"2105.04607","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-free-gradient-temporal-difference","title":"Parameter-free Gradient Temporal Difference Learning","date":"2021-05-10","arxiv_id":"2105.04129","repositories_listed":0,"syntology":null},{"url":null,"slug":"pearl-parallelized-expert-assisted","title":"PEARL: Parallelized Expert-Assisted Reinforcement Learning for Scene Rearrangement Planning","date":"2021-05-10","arxiv_id":"2105.04088","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-rare-diffusive","title":"Reinforcement learning of rare diffusive dynamics","date":"2021-05-10","arxiv_id":"2105.04321","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-cost-learning-for-jpeg","title":"Improving Cost Learning for JPEG Steganography by Exploiting JPEG Domain Knowledge","date":"2021-05-09","arxiv_id":"2105.03867","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-expert-trajectory","title":"Reinforcement Learning with Expert Trajectory For Quantitative Trading","date":"2021-05-09","arxiv_id":"2105.03844","repositories_listed":0,"syntology":null},{"url":null,"slug":"mctg-multi-frequency-continuous-share-trading","title":"A parallel-network continuous quantitative trading model with GARCH and PPO","date":"2021-05-08","arxiv_id":"2105.03625","repositories_listed":0,"syntology":null},{"url":null,"slug":"rail-a-modular-framework-for-reinforcement","title":"RAIL: A modular framework for Reinforcement-learning-based Adversarial Imitation Learning","date":"2021-05-08","arxiv_id":"2105.03756","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-decentralized-multi-agent","title":"Scalable, Decentralized Multi-Agent Reinforcement Learning Methods Inspired by Stigmergy and Ant Colonies","date":"2021-05-08","arxiv_id":"2105.03546","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-prediction-for-representation-learning","title":"Reward prediction for representation learning and reward shaping","date":"2021-05-07","arxiv_id":"2105.03172","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-reinforcement-learning-to-design-an-ai","title":"Using reinforcement learning to design an AI assistantfor a satisfying co-op experience","date":"2021-05-07","arxiv_id":"2105.03414","repositories_listed":0,"syntology":null},{"url":null,"slug":"utilizing-skipped-frames-in-action-repeats","title":"Utilizing Skipped Frames in Action Repeats via Pseudo-Actions","date":"2021-05-07","arxiv_id":"2105.03041","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-economic-model","title":"A Reinforcement Learning-based Economic Model Predictive Control Framework for Autonomous Operation of Chemical Reactors","date":"2021-05-06","arxiv_id":"2105.02656","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-graph-convolutional-reinforcement","title":"Deep Graph Convolutional Reinforcement Learning for Financial Portfolio Management -- DeepPocket","date":"2021-05-06","arxiv_id":"2105.08664","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-aware-q-networks-resolving-temporal","title":"Time-Aware Q-Networks: Resolving Temporal Irregularity for Deep Reinforcement Learning","date":"2021-05-06","arxiv_id":"2105.02580","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-algorithms-for-regenerative-stopping","title":"Learning Algorithms for Regenerative Stopping Problems with Applications to Shipping Consolidation in Logistics","date":"2021-05-05","arxiv_id":"2105.02318","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-enhancement-for-deep-reinforcement","title":"Safety Enhancement for Deep Reinforcement Learning in Autonomous Separation Assurance","date":"2021-05-05","arxiv_id":"2105.02331","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-sokoban-with-backward-reinforcement","title":"Solving Sokoban with forward-backward reinforcement learning","date":"2021-05-05","arxiv_id":"2105.01904","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-on-multi-agent-q-learning-frameworks","title":"Survey on Multi-Agent Q-Learning frameworks for resource management in wireless sensor network","date":"2021-05-05","arxiv_id":"2105.02371","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-reinforcement-learning-for-1","title":"Data-Efficient Reinforcement Learning for Malaria Control","date":"2021-05-04","arxiv_id":"2105.01620","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-lottery-tickets-and-minimal-task","title":"On Lottery Tickets and Minimal Task Representations in Deep Reinforcement Learning","date":"2021-05-04","arxiv_id":"2105.01648","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-linear-convergence-of-natural-policy","title":"On the Linear convergence of Natural Policy Gradient Algorithm","date":"2021-05-04","arxiv_id":"2105.01424","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-scalable-logic","title":"Reinforcement Learning for Scalable Logic Optimization with Graph Neural Networks","date":"2021-05-04","arxiv_id":"2105.01755","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-adversarial-reward-learning-for","title":"Generative Adversarial Reward Learning for Generalized Behavior Tendency Inference","date":"2021-05-03","arxiv_id":"2105.00822","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-air","title":"Hierarchical Reinforcement Learning for Air-to-Air Combat","date":"2021-05-03","arxiv_id":"2105.00990","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-swimming-escape-patterns-under","title":"Learning swimming escape patterns for larval fish under energy constraints","date":"2021-05-03","arxiv_id":"2105.00771","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-ridesharing-a","title":"Reinforcement Learning for Ridesharing: An Extended Survey","date":"2021-05-03","arxiv_id":"2105.01099","repositories_listed":0,"syntology":null},{"url":null,"slug":"backdoorl-backdoor-attack-against-competitive","title":"BACKDOORL: Backdoor Attack against Competitive Reinforcement Learning","date":"2021-05-02","arxiv_id":"2105.00579","repositories_listed":0,"syntology":null},{"url":null,"slug":"carl-dtn-context-adaptive-reinforcement","title":"CARL-DTN: Context Adaptive Reinforcement Learning based Routing Algorithm in Delay Tolerant Network","date":"2021-05-02","arxiv_id":"2105.00544","repositories_listed":0,"syntology":null},{"url":null,"slug":"infernet-for-delayed-reinforcement-tasks","title":"InferNet for Delayed Reinforcement Tasks: Addressing the Temporal Credit Assignment Problem","date":"2021-05-02","arxiv_id":"2105.00568","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-bus-bunching-with-asynchronous-multi","title":"Reducing Bus Bunching with Asynchronous Multi-Agent Reinforcement Learning","date":"2021-05-02","arxiv_id":"2105.00376","repositories_listed":0,"syntology":null},{"url":null,"slug":"better-than-the-best-gradient-based-improper","title":"Better than the Best: Gradient-based Improper Reinforcement Learning for Network Scheduling","date":"2021-05-01","arxiv_id":"2105.00210","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-swarm-collision-avoidance-for","title":"Nearest-Neighbor-based Collision Avoidance for Quadrotors via Reinforcement Learning","date":"2021-04-30","arxiv_id":"2104.14912","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-time-mean-field-control-with","title":"Discrete-Time Mean Field Control with Environment States","date":"2021-04-30","arxiv_id":"2104.14900","repositories_listed":0,"syntology":null},{"url":null,"slug":"mean-field-marl-based-bandwidth-negotiation","title":"Mean Field MARL Based Bandwidth Negotiation Method for Massive Devices Spectrum Sharing","date":"2021-04-30","arxiv_id":"2104.15085","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-political-bias-in-language-models","title":"Mitigating Political Bias in Language Models Through Reinforced Calibration","date":"2021-04-30","arxiv_id":"2104.14795","repositories_listed":0,"syntology":null},{"url":null,"slug":"antagonistic-crowd-simulation-model","title":"Emotional Contagion-Aware Deep Reinforcement Learning for Antagonistic Crowd Simulation","date":"2021-04-29","arxiv_id":"2105.00854","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypernetwork-dismantling-via-deep","title":"Hypernetwork Dismantling via Deep Reinforcement Learning","date":"2021-04-29","arxiv_id":"2104.14332","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-entropy-inverse-reinforcement","title":"Adversarial Inverse Reinforcement Learning for Mean Field Games","date":"2021-04-29","arxiv_id":"2104.14654","repositories_listed":0,"syntology":null},{"url":null,"slug":"medium-access-using-distributed-reinforcement","title":"Medium Access using Distributed Reinforcement Learning for IoTs with Low-Complexity Wireless Transceivers","date":"2021-04-29","arxiv_id":"2104.14549","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-of-deep-rl-agents-for-improved","title":"Pre-training of Deep RL Agents for Improved Learning under Domain Randomization","date":"2021-04-29","arxiv_id":"2104.14386","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-meta-reinforcement-learning-to-bridge","title":"Using Meta Reinforcement Learning to Bridge the Gap between Simulation and Experiment in Energy Demand Response","date":"2021-04-29","arxiv_id":"2104.14670","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-is-going-on-inside-recurrent-meta","title":"What is Going on Inside Recurrent Meta Reinforcement Learning Agents?","date":"2021-04-29","arxiv_id":"2104.14644","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-generalized-projected-bellman-error-for-off","title":"A Generalized Projected Bellman Error for Off-policy Value Estimation in Reinforcement Learning","date":"2021-04-28","arxiv_id":"2104.13844","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-environment-for","title":"A Reinforcement Learning Environment for Polyhedral Optimizations","date":"2021-04-28","arxiv_id":"2104.13732","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-intersection-handling-using-multi","title":"End-to-End Intersection Handling using Multi-Agent Deep Reinforcement Learning","date":"2021-04-28","arxiv_id":"2104.13617","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-mis-design-for-autonomous-driving","title":"Reward (Mis)design for Autonomous Driving","date":"2021-04-28","arxiv_id":"2104.13906","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-adversarial-training-for-meta","title":"Adaptive Adversarial Training for Meta Reinforcement Learning","date":"2021-04-27","arxiv_id":"2104.13302","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-earthquake-like-instabilities","title":"Controlling earthquake-like instabilities using artificial intelligence","date":"2021-04-27","arxiv_id":"2104.13180","repositories_listed":0,"syntology":null},{"url":null,"slug":"implementing-reinforcement-learning","title":"Implementing Reinforcement Learning Algorithms in Retail Supply Chains with OpenAI Gym Toolkit","date":"2021-04-27","arxiv_id":"2104.14398","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-on-policy-training-for-sample-efficient","title":"Semi-On-Policy Training for Sample Efficient Multi-Agent Policy Gradients","date":"2021-04-27","arxiv_id":"2104.13446","repositories_listed":0,"syntology":null},{"url":null,"slug":"ant-learning-accurate-network-throughput-for","title":"ANT: Learning Accurate Network Throughput for Better Adaptive Video Streaming","date":"2021-04-26","arxiv_id":"2104.12507","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-5","title":"A Deep Reinforcement Learning Approach for the Meal Delivery Problem","date":"2021-04-24","arxiv_id":"2104.12000","repositories_listed":0,"syntology":null},{"url":null,"slug":"disco-rl-distribution-conditioned","title":"DisCo RL: Distribution-Conditioned Reinforcement Learning for General-Purpose Policies","date":"2021-04-23","arxiv_id":"2104.11707","repositories_listed":0,"syntology":null},{"url":null,"slug":"formula-rl-deep-reinforcement-learning-for","title":"Formula RL: Deep Reinforcement Learning for Autonomous Racing using Telemetry Data","date":"2021-04-22","arxiv_id":"2104.11106","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-using-guided","title":"Reinforcement Learning using Guided Observability","date":"2021-04-22","arxiv_id":"2104.10986","repositories_listed":0,"syntology":null},{"url":null,"slug":"reset-free-reinforcement-learning-via-multi","title":"Reset-Free Reinforcement Learning via Multi-Task Learning: Learning Dexterous Manipulation Behaviors without Human Intervention","date":"2021-04-22","arxiv_id":"2104.11203","repositories_listed":0,"syntology":null},{"url":null,"slug":"cvlight-deep-reinforcement-learning-for","title":"CVLight: Decentralized Learning for Adaptive Traffic Signal Control with Connected Vehicles","date":"2021-04-21","arxiv_id":"2104.10340","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-aided-deep-reinforcement-learning-for","title":"Model-aided Deep Reinforcement Learning for Sample-efficient UAV Trajectory Design in IoT Networks","date":"2021-04-21","arxiv_id":"2104.10403","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-fusion-for-adaptive-and-customizable","title":"Policy Fusion for Adaptive and Customizable Reinforcement Learning Agents","date":"2021-04-21","arxiv_id":"2104.10610","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-traffic-signal","title":"Reinforcement Learning for Traffic Signal Control: Comparison with Commercial Systems","date":"2021-04-21","arxiv_id":"2104.10455","repositories_listed":0,"syntology":null},{"url":null,"slug":"tackling-variabilities-in-autonomous-driving","title":"Tackling Variabilities in Autonomous Driving","date":"2021-04-21","arxiv_id":"2104.10415","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-an-aid-policy-to-minimize-student","title":"Discovering an Aid Policy to Minimize Student Evasion Using Offline Reinforcement Learning","date":"2021-04-20","arxiv_id":"2104.10258","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-deep-reinforcement-learning-for","title":"DRL: Deep Reinforcement Learning for Intelligent Robot Control -- Concept, Literature, and Future","date":"2021-04-20","arxiv_id":"2105.13806","repositories_listed":0,"syntology":null},{"url":null,"slug":"glide-generalizable-quadrupedal-locomotion-in","title":"GLiDE: Generalizable Quadrupedal Locomotion in Diverse Environments with a Centroidal Model","date":"2021-04-20","arxiv_id":"2104.09771","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-predictive-control-and-reinforcement","title":"Model-predictive control and reinforcement learning in multi-energy system case studies","date":"2021-04-20","arxiv_id":"2104.09785","repositories_listed":0,"syntology":null},{"url":null,"slug":"network-defense-is-not-a-game","title":"Network Defense is Not a Game","date":"2021-04-20","arxiv_id":"2104.10262","repositories_listed":0,"syntology":null},{"url":null,"slug":"network-wide-traffic-signal-control","title":"Network-wide traffic signal control optimization using a multi-agent deep reinforcement learning","date":"2021-04-20","arxiv_id":"2104.09936","repositories_listed":0,"syntology":null},{"url":null,"slug":"outcome-driven-reinforcement-learning-via","title":"Outcome-Driven Reinforcement Learning via Variational Inference","date":"2021-04-20","arxiv_id":"2104.10190","repositories_listed":0,"syntology":null},{"url":null,"slug":"prospective-artificial-intelligence","title":"Prospective Artificial Intelligence Approaches for Active Cyber Defence","date":"2021-04-20","arxiv_id":"2104.09981","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-synthesis-of-verified-controllers-in","title":"Scalable Synthesis of Verified Controllers in Deep Reinforcement Learning","date":"2021-04-20","arxiv_id":"2104.10219","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-learning-for-financial-markets","title":"Adaptive learning for financial markets mixing model-based and model-free RL for volatility targeting","date":"2021-04-19","arxiv_id":"2104.10483","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-centric-representations-for-multi-agent","title":"Agent-Centric Representations for Multi-Agent Reinforcement Learning","date":"2021-04-19","arxiv_id":"2104.09402","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-multi-agent-fitted-q-iteration","title":"Approximated Multi-Agent Fitted Q Iteration","date":"2021-04-19","arxiv_id":"2104.09343","repositories_listed":0,"syntology":null},{"url":null,"slug":"constraints-satisfiability-driven","title":"Constraints Satisfiability Driven Reinforcement Learning for Autonomous Cyber Defense","date":"2021-04-19","arxiv_id":"2104.08994","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-a-monetary","title":"Deep Reinforcement Learning in a Monetary Model","date":"2021-04-19","arxiv_id":"2104.09368","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-linear-convex","title":"Reinforcement learning for linear-convex models with jumps via stability analysis of feedback controls","date":"2021-04-19","arxiv_id":"2104.09311","repositories_listed":0,"syntology":null},{"url":null,"slug":"singular-perturbation-based-reinforcement","title":"Singular Perturbation-based Reinforcement Learning of Two-Point Boundary Optimal Control Systems","date":"2021-04-19","arxiv_id":"2104.09652","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-value-aligned-reinforcement-learning","title":"Training Value-Aligned Reinforcement Learning Agents Using a Normative Prior","date":"2021-04-19","arxiv_id":"2104.09469","repositories_listed":0,"syntology":null},{"url":null,"slug":"mt-opt-continuous-multi-task-robotic","title":"MT-Opt: Continuous Multi-Task Robotic Reinforcement Learning at Scale","date":"2021-04-16","arxiv_id":"2104.08212","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-exploration-in-model-based-reinforcement","title":"Safe Exploration in Model-based Reinforcement Learning using Control Barrier Functions","date":"2021-04-16","arxiv_id":"2104.08171","repositories_listed":0,"syntology":null},{"url":null,"slug":"actionable-models-unsupervised-offline","title":"Actionable Models: Unsupervised Offline Reinforcement Learning of Robotic Skills","date":"2021-04-15","arxiv_id":"2104.07749","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-l-2-analysis-of-reinforcement-learning-in","title":"An $L^2$ Analysis of Reinforcement Learning in High Dimensions with Kernel and Neural Network Approximation","date":"2021-04-15","arxiv_id":"2104.07794","repositories_listed":0,"syntology":null},{"url":null,"slug":"discover-the-hidden-attack-path-in-multi","title":"Discover the Hidden Attack Path in Multi-domain Cyberspace Based on Reinforcement Learning","date":"2021-04-15","arxiv_id":"2104.07195","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-attention-for-multi-agent-coordination","title":"Joint Attention for Multi-Agent Coordination and Social Learning","date":"2021-04-15","arxiv_id":"2104.07750","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-based-3","title":"Multi-Agent Reinforcement Learning Based Coded Computation for Mobile Ad Hoc Computing","date":"2021-04-15","arxiv_id":"2104.07539","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictor-corrector-pc-temporal-difference-td","title":"Predictor-Corrector(PC) Temporal Difference(TD) Learning (PCTD)","date":"2021-04-15","arxiv_id":"2104.09620","repositories_listed":0,"syntology":null}],"record_sha256":"f7d010749c55a4161ede9f5aa3178322fecbdaa57b950cab991e28637de235d4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}