{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/124","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":124,"pages_in_order":152,"rows_per_page":100,"rows":[12301,12400],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/123","next":"/task/reinforcement-learning-1/papers/125","papers":[{"url":null,"slug":"qstar-approximation-schemes-for-batch","title":"Q* Approximation Schemes for Batch Reinforcement Learning: A Theoretical Comparison","date":"2020-03-09","arxiv_id":"2003.03924","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-reinforcement-learning-under","title":"Transfer Reinforcement Learning under Unobserved Contextual Information","date":"2020-03-09","arxiv_id":"2003.04427","repositories_listed":0,"syntology":null},{"url":null,"slug":"zooming-for-efficient-model-free","title":"Zooming for Efficient Model-Free Reinforcement Learning in Metric Spaces","date":"2020-03-09","arxiv_id":"2003.04069","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-adversarial-reinforcement-learning-for","title":"Deep Adversarial Reinforcement Learning for Object Disentangling","date":"2020-03-08","arxiv_id":"2003.03779","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-adversarial-imitation-learning-2","title":"Generative Adversarial Imitation Learning with Neural Networks: Global Optimality and Convergence Rate","date":"2020-03-08","arxiv_id":"2003.03709","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-cooperative","title":"Reinforcement Learning Based Cooperative Coded Caching under Dynamic Popularities in Ultra-Dense Networks","date":"2020-03-08","arxiv_id":"2003.03758","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-of-q-value-in-case-of-gaussian","title":"Convergence of Q-value in case of Gaussian rewards","date":"2020-03-07","arxiv_id":"2003.03526","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-combinatorial","title":"Reinforcement Learning for Combinatorial Optimization: A Survey","date":"2020-03-07","arxiv_id":"2003.03600","repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-sensitive-portfolio-selection-via-deep","title":"Cost-Sensitive Portfolio Selection via Deep Reinforcement Learning","date":"2020-03-06","arxiv_id":"2003.03051","repositories_listed":0,"syntology":null},{"url":null,"slug":"lane-merging-using-policy-based-reinforcement","title":"Lane-Merging Using Policy-based Reinforcement Learning and Post-Optimization","date":"2020-03-06","arxiv_id":"2003.03168","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-train-operation-algorithms-based-on","title":"Smart Train Operation Algorithms based on Expert Knowledge and Reinforcement Learning","date":"2020-03-06","arxiv_id":"2003.03327","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-geometric-perspective-on-visual-imitation","title":"A Geometric Perspective on Visual Imitation Learning","date":"2020-03-05","arxiv_id":"2003.02768","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-robust","title":"Deep Reinforcement Learning-BasedRobust Protection in DER-Rich Distribution Grids","date":"2020-03-05","arxiv_id":"2003.02422","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-robustness-and-regularization","title":"Distributional Robustness and Regularization in Reinforcement Learning","date":"2020-03-05","arxiv_id":"2003.02894","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-and-effective-similar-subtrajectory","title":"Efficient and Effective Similar Subtrajectory Search with Deep Reinforcement Learning","date":"2020-03-05","arxiv_id":"2003.02542","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-design-in-cooperative-multi-agent-1","title":"Reward Design in Cooperative Multi-agent Reinforcement Learning for Packet Routing","date":"2020-03-05","arxiv_id":"2003.03433","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-experience-replay","title":"Dynamic Experience Replay","date":"2020-03-04","arxiv_id":"2003.02372","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-statistical-validation-with-edge","title":"Efficient statistical validation with edge cases to evaluate Highly Automated Vehicles","date":"2020-03-04","arxiv_id":"2003.01886","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-network-heuristics-for-adaptive","title":"Neural-Network Heuristics for Adaptive Bayesian Quantum Estimation","date":"2020-03-04","arxiv_id":"2003.02183","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-aware-time-series-data-sharing-with","title":"Privacy-Aware Time-Series Data Sharing with Deep Reinforcement Learning","date":"2020-03-04","arxiv_id":"2003.02685","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-qos","title":"Deep Reinforcement Learning for QoS-Constrained Resource Allocation in Multiservice Networks","date":"2020-03-03","arxiv_id":"2003.02643","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-exploration-in-constrained","title":"Efficient Exploration in Constrained Environments with Goal-Oriented Reference Path","date":"2020-03-03","arxiv_id":"2003.01641","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-context-aware-task-reasoning-for","title":"Learning Context-aware Task Reasoning for Efficient Meta-reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01373","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-autonomous","title":"Safe Reinforcement Learning for Autonomous Vehicles through Parallel Constrained Policy Optimization","date":"2020-03-03","arxiv_id":"2003.01303","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-object-level-deep","title":"Relevance-Guided Modeling of Object Dynamics for Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01384","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-structural-hyper-parameter","title":"Adaptive Structural Hyper-Parameter Configuration by Q-Learning","date":"2020-03-02","arxiv_id":"2003.00863","repositories_listed":0,"syntology":null},{"url":null,"slug":"cluster-based-social-reinforcement-learning","title":"Cluster-Based Social Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.00627","repositories_listed":0,"syntology":null},{"url":null,"slug":"formal-controller-synthesis-for-continuous","title":"Formal Controller Synthesis for Continuous-Space MDPs via Model-Free Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.00712","repositories_listed":0,"syntology":null},{"url":null,"slug":"gaussian-process-policy-optimization","title":"Gaussian Process Policy Optimization","date":"2020-03-02","arxiv_id":"2003.01074","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-contact-rich-manipulation-tasks-with","title":"Learning Force Control for Contact-rich Manipulation Tasks with Rigid Position-controlled Robots","date":"2020-03-02","arxiv_id":"2003.00628","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-human-robot-collaborative","title":"Real-World Human-Robot Collaborative Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.01156","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-averse-learning-by-temporal-difference","title":"Risk-Averse Learning by Temporal Difference Methods","date":"2020-03-02","arxiv_id":"2003.00780","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-up-multiagent-reinforcement-learning","title":"Scaling Up Multiagent Reinforcement Learning for Robotic Systems: Learn an Adaptive Sparse Communication Graph","date":"2020-03-02","arxiv_id":"2003.01040","repositories_listed":0,"syntology":null},{"url":null,"slug":"upper-confidence-primal-dual-optimization","title":"Upper Confidence Primal-Dual Reinforcement Learning for CMDP with Adversarial Loss","date":"2020-03-02","arxiv_id":"2003.00660","repositories_listed":0,"syntology":null},{"url":null,"slug":"v2i-connectivity-based-dynamic-queue-jumper","title":"Dynamic Queue-Jump Lane for Emergency Vehicles under Partially Connected Settings: A Multi-Agent Deep Reinforcement Learning Approach","date":"2020-03-02","arxiv_id":"2003.01025","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-policy-evaluation-in-distributed","title":"Fully Asynchronous Policy Evaluation in Distributed Reinforcement Learning over Networks","date":"2020-03-01","arxiv_id":"2003.00433","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-policy-reuse-using-deep-mixture","title":"Contextual Policy Transfer in Reinforcement Learning Domains via Deep Mixtures-of-Experts","date":"2020-02-29","arxiv_id":"2003.00203","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-near-optimal-policies-with-low","title":"Learning Near Optimal Policies with Low Inherent Bellman Error","date":"2020-02-29","arxiv_id":"2003.00153","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixed-reinforcement-learning-with-additive","title":"Mixed Reinforcement Learning with Additive Stochastic Uncertainty","date":"2020-02-28","arxiv_id":"2003.00848","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-flipit","title":"Deep Reinforcement Learning for FlipIt Security Game","date":"2020-02-28","arxiv_id":"2002.12909","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-through-active","title":"Reinforcement Learning through Active Inference","date":"2020-02-28","arxiv_id":"2002.12636","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-tuning-deep-reinforcement-learning","title":"A Self-Tuning Actor-Critic Algorithm","date":"2020-02-28","arxiv_id":"2002.12928","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-visual-communication-map-for-multi-agent","title":"A Visual Communication Map for Multi-Agent Deep Reinforcement Learning","date":"2020-02-27","arxiv_id":"2002.11882","repositories_listed":0,"syntology":null},{"url":null,"slug":"acceleration-of-actor-critic-deep","title":"Acceleration of Actor-Critic Deep Reinforcement Learning for Visual Grasping in Clutter by State Representation Learning Based on Disentanglement of a Raw Input Image","date":"2020-02-27","arxiv_id":"2002.11903","repositories_listed":0,"syntology":null},{"url":null,"slug":"assembly-robots-with-optimized-control","title":"Assembly robots with optimized control stiffness through reinforcement learning","date":"2020-02-27","arxiv_id":"2002.12207","repositories_listed":0,"syntology":null},{"url":null,"slug":"cautious-reinforcement-learning-via","title":"Cautious Reinforcement Learning via Distributional Risk in the Dual Domain","date":"2020-02-27","arxiv_id":"2002.12475","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-intelligent","title":"Deep Reinforcement Learning Based Intelligent Reflecting Surface for Secure Wireless Communications","date":"2020-02-27","arxiv_id":"2002.12271","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-in-markov-decision-processes-under","title":"Learning in Markov Decision Processes under Constraints","date":"2020-02-27","arxiv_id":"2002.12435","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-resolve-alliance-dilemmas-in-many","title":"Learning to Resolve Alliance Dilemmas in Many-Player Zero-Sum Games","date":"2020-02-27","arxiv_id":"2003.00799","repositories_listed":0,"syntology":null},{"url":null,"slug":"review-analyze-and-design-a-comprehensive","title":"Review, Analysis and Design of a Comprehensive Deep Reinforcement Learning Framework","date":"2020-02-27","arxiv_id":"2002.11883","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-goal-trees-a-framework-for-goal-based","title":"Sub-Goal Trees -- a Framework for Goal-Based Reinforcement Learning","date":"2020-02-27","arxiv_id":"2002.12361","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-modular-algorithm-induction","title":"Towards Modular Algorithm Induction","date":"2020-02-27","arxiv_id":"2003.04227","repositories_listed":0,"syntology":null},{"url":null,"slug":"cautious-reinforcement-learning-with-logical","title":"Cautious Reinforcement Learning with Logical Constraints","date":"2020-02-26","arxiv_id":"2002.12156","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-hindsight-for-reinforcement","title":"Generalized Hindsight for Reinforcement Learning","date":"2020-02-26","arxiv_id":"2002.11708","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-ordinary-differential-equation-value","title":"Neural Ordinary Differential Equation Value Networks for Parametrized Action Spaces","date":"2020-02-26","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"when-do-drivers-concentrate-attention-based","title":"When Do Drivers Concentrate? Attention-based Driver Behavior Modeling With Deep Reinforcement Learning","date":"2020-02-26","arxiv_id":"2002.11385","repositories_listed":0,"syntology":null},{"url":null,"slug":"g-learner-and-girl-goal-based-wealth","title":"G-Learner and GIRL: Goal Based Wealth Management with Reinforcement Learning","date":"2020-02-25","arxiv_id":"2002.10990","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-2","title":"Model-Based Reinforcement Learning for Physical Systems Without Velocity and Acceleration Measurements","date":"2020-02-25","arxiv_id":"2002.10621","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-reinforcement-learning-for-turn-based-zero","title":"On Reinforcement Learning for Turn-based Zero-sum Markov Games","date":"2020-02-25","arxiv_id":"2002.10620","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-multi-task-imitation-learning-with","title":"Scalable Multi-Task Imitation Learning with Autonomous Improvement","date":"2020-02-25","arxiv_id":"2003.02636","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneously-evolving-deep-reinforcement","title":"Simultaneously Evolving Deep Reinforcement Learning Models using Multifactorial Optimization","date":"2020-02-25","arxiv_id":"2002.12133","repositories_listed":0,"syntology":null},{"url":null,"slug":"backpropamine-training-self-modifying-neural-1","title":"Backpropamine: training self-modifying neural networks with differentiable neuromodulated plasticity","date":"2020-02-24","arxiv_id":"2002.10585","repositories_listed":0,"syntology":null},{"url":null,"slug":"millimeter-wave-communications-with-an","title":"Millimeter Wave Communications with an Intelligent Reflector: Performance Optimization and Distributional Reinforcement Learning","date":"2020-02-24","arxiv_id":"2002.10572","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-multi-agent-inverse-reinforcement","title":"Scalable Multi-Agent Inverse Reinforcement Learning via Actor-Attention-Critic","date":"2020-02-24","arxiv_id":"2002.10525","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-linear","title":"Deep Reinforcement Learning with Linear Quadratic Regulator Regions","date":"2020-02-23","arxiv_id":"2002.09820","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-regret-bounds-for-stochastic","title":"Near-optimal Regret Bounds for Stochastic Shortest Path","date":"2020-02-23","arxiv_id":"2002.09869","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-traffic-lights-with-multi-agent","title":"Optimizing Traffic Lights with Multi-agent Deep Reinforcement Learning and V2X communication","date":"2020-02-23","arxiv_id":"2002.09853","repositories_listed":0,"syntology":null},{"url":null,"slug":"rapidly-personalizing-mobile-health-treatment","title":"Rapidly Personalizing Mobile Health Treatment Policies with Limited Data","date":"2020-02-23","arxiv_id":"2002.09971","repositories_listed":0,"syntology":null},{"url":null,"slug":"wireless-20-towards-an-intelligent-radio","title":"Wireless 2.0: Towards an Intelligent Radio Environment Empowered by Reconfigurable Meta-Surfaces and Artificial Intelligence","date":"2020-02-23","arxiv_id":"2002.11040","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-radar-inference-from-inverse","title":"Adversarial Radar Inference. From Inverse Tracking to Inverse Reinforcement Learning of Cognitive Radar","date":"2020-02-22","arxiv_id":"2002.10910","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-data-augmentation-via-deep","title":"Automatic Data Augmentation via Deep Reinforcement Learning for Effective Kidney Tumor Segmentation","date":"2020-02-22","arxiv_id":"2002.09703","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-constrained-policy-optimization-for","title":"Guided Constrained Policy Optimization for Dynamic Quadrupedal Robot Locomotion","date":"2020-02-22","arxiv_id":"2002.09676","repositories_listed":0,"syntology":null},{"url":null,"slug":"vehicle-tracking-in-wireless-sensor-networks","title":"Vehicle Tracking in Wireless Sensor Networks via Deep Reinforcement Learning","date":"2020-02-22","arxiv_id":"2002.09671","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-reinforcement-learning-with-a","title":"Accelerating Reinforcement Learning with a Directional-Gaussian-Smoothing Evolution Strategy","date":"2020-02-21","arxiv_id":"2002.09077","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-freshness-and-energy-efficient-uav","title":"Data Freshness and Energy-Efficient UAV Navigation Optimization: A Deep Reinforcement Learning Approach","date":"2020-02-21","arxiv_id":"2003.04816","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-controllable-object-through","title":"Disentangling Controllable Object through Video Prediction Improves Visual Reinforcement Learning","date":"2020-02-21","arxiv_id":"2002.09136","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-search-for-feedback-in-reinforcement","title":"On the Search for Feedback in Reinforcement Learning","date":"2020-02-21","arxiv_id":"2002.09478","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-temporal-difference-learning-with","title":"Adaptive Temporal Difference Learning with Linear Function Approximation","date":"2020-02-20","arxiv_id":"2002.08537","repositories_listed":0,"syntology":null},{"url":"/paper/automatic-gesture-recognition-in-robot","slug":"automatic-gesture-recognition-in-robot","title":"Automatic Gesture Recognition in Robot-assisted Surgery with Reinforcement Learning and Tree Search","date":"2020-02-20","arxiv_id":"2002.08718","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-adversarial-strategically-timed","title":"Enhanced Adversarial Strategically-Timed Attacks against Deep Reinforcement Learning","date":"2020-02-20","arxiv_id":"2002.09027","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-meta-reinforcement-learning-for","title":"Multi-Agent Meta-Reinforcement Learning for Self-Powered and Sustainable Edge Computing Systems","date":"2020-02-20","arxiv_id":"2002.08567","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-as-a","title":"Multi-Agent Reinforcement Learning as a Computational Tool for Language Evolution Research: Historical Context and Future Challenges","date":"2020-02-20","arxiv_id":"2002.08878","repositories_listed":0,"syntology":null},{"url":null,"slug":"oirl-robust-adversarial-inverse-reinforcement","title":"oIRL: Robust Adversarial Inverse Reinforcement Learning with Temporally Extended Actions","date":"2020-02-20","arxiv_id":"2002.09043","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-counterfactual-reinforcement-learning","title":"Debiased Off-Policy Evaluation for Recommendation Systems","date":"2020-02-20","arxiv_id":"2002.08536","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-in-gradient-based-meta","title":"Curriculum in Gradient-Based Meta-Reinforcement Learning","date":"2020-02-19","arxiv_id":"2002.07956","repositories_listed":0,"syntology":null},{"url":null,"slug":"keep-doing-what-worked-behavioral-modelling","title":"Keep Doing What Worked: Behavioral Modelling Priors for Offline Reinforcement Learning","date":"2020-02-19","arxiv_id":"2002.08396","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimistic-policy-optimization-with-bandit","title":"Optimistic Policy Optimization with Bandit Feedback","date":"2020-02-19","arxiv_id":"2002.08243","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-aided-search-and-rescue-operation-using","title":"UAV Aided Search and Rescue Operation Using Reinforcement Learning","date":"2020-02-19","arxiv_id":"2002.08415","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-driven-hindsight-modelling-1","title":"Value-driven Hindsight Modelling","date":"2020-02-19","arxiv_id":"2002.08329","repositories_listed":0,"syntology":null},{"url":null,"slug":"empirical-policy-evaluation-with-supergraphs","title":"Empirical Policy Evaluation with Supergraphs","date":"2020-02-18","arxiv_id":"2002.07905","repositories_listed":0,"syntology":null},{"url":null,"slug":"kogun-accelerating-deep-reinforcement","title":"KoGuN: Accelerating Deep Reinforcement Learning via Integrating Human Suboptimal Knowledge","date":"2020-02-18","arxiv_id":"2002.07418","repositories_listed":0,"syntology":null},{"url":null,"slug":"motiac-multi-objective-actor-critics-for-real","title":"MoTiAC: Multi-Objective Actor-Critics for Real-Time Bidding","date":"2020-02-18","arxiv_id":"2002.07408","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-issue-bargaining-with-deep","title":"Multi-Issue Bargaining With Deep Reinforcement Learning","date":"2020-02-18","arxiv_id":"2002.07788","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-experience-selection-for-policy","title":"Adaptive Experience Selection for Policy Gradient","date":"2020-02-17","arxiv_id":"2002.06946","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-the-manipulation","title":"Reinforcement learning for the privacy preservation and manipulation of eye tracking data","date":"2020-02-17","arxiv_id":"2002.06806","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-design-for-driver-repositioning-using","title":"Reward Design for Driver Repositioning Using Multi-Agent Reinforcement Learning","date":"2020-02-17","arxiv_id":"2002.06723","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-simple-object-representations","title":"Investigating Simple Object Representations in Model-Free Deep Reinforcement Learning","date":"2020-02-16","arxiv_id":"2002.06703","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-asymptotic-convergence-of-adam-type","title":"Non-asymptotic Convergence of Adam-type Reinforcement Learning Algorithms under Markovian Sampling","date":"2020-02-15","arxiv_id":"2002.06286","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-archimedean-trap-why-traditional","title":"The Archimedean trap: Why traditional reinforcement learning will probably not yield AGI","date":"2020-02-15","arxiv_id":"2002.10221","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-management-in-wireless-networks-via","title":"Resource Management in Wireless Networks via Multi-Agent Deep Reinforcement Learning","date":"2020-02-14","arxiv_id":"2002.06215","repositories_listed":0,"syntology":null}],"record_sha256":"7c1126692df17d89b1ff44076af205f16d8ef85e0b884eb653aeac8f764474ea","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}