{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/114","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":114,"pages_in_order":135,"rows_per_page":100,"rows":[11301,11400],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/113","next":"/task/reinforcement-learning-2/papers/115","papers":[{"url":null,"slug":"dynamic-pricing-on-e-commerce-platform-with","title":"Dynamic Pricing on E-commerce Platform with Deep Reinforcement Learning: A Field Experiment","date":"2019-12-05","arxiv_id":"1912.02572","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-policy-space-expansion-in","title":"Iterative Policy-Space Expansion in Reinforcement Learning","date":"2019-12-05","arxiv_id":"1912.02532","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-convolutional","title":"Reinforcement Learning with Convolutional Reservoir Computing","date":"2019-12-05","arxiv_id":"1912.04161","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-reinforcement-learning-of-localized","title":"Scalable Reinforcement Learning for Multi-Agent Networked Systems","date":"2019-12-05","arxiv_id":"1912.02906","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-model-compression-via-deep-reinforcement","title":"Deep Model Compression Via Two-Stage Deep Reinforcement Learning","date":"2019-12-04","arxiv_id":"1912.02254","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-bandwidth","title":"Reinforcement learning for bandwidth estimation and congestion control in real-time communications","date":"2019-12-04","arxiv_id":"1912.02222","repositories_listed":0,"syntology":null},{"url":null,"slug":"neighborhood-cognition-consistent-multi-agent","title":"Neighborhood Cognition Consistent Multi-Agent Reinforcement Learning","date":"2019-12-03","arxiv_id":"1912.01160","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-learned-formula-synthesis-in-set-theory","title":"Self-Learned Formula Synthesis in Set Theory","date":"2019-12-03","arxiv_id":"1912.01525","repositories_listed":0,"syntology":null},{"url":null,"slug":"191201715","title":"Human-Robot Collaboration via Deep Reinforcement Learning of Real-World Interactions","date":"2019-12-02","arxiv_id":"1912.01715","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-policy-reinforcement-learning-with-entropy","title":"Policy Optimization Reinforcement Learning with Entropy Regularization","date":"2019-12-02","arxiv_id":"1912.01557","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-family-of-robust-stochastic-operators-for","title":"A Family of Robust Stochastic Operators for Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adversary-a3c-for-robust-reinforcement-1","title":"Adversary A3C for Robust Reinforcement Learning","date":"2019-12-01","arxiv_id":"1912.00330","repositories_listed":0,"syntology":null},{"url":null,"slug":"explicit-planning-for-efficient-exploration","title":"Explicit Planning for Efficient Exploration in Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"flow-rate-control-in-smart-district-heating","title":"Flow Rate Control in Smart District Heating Systems Using Deep Reinforcement Learning","date":"2019-12-01","arxiv_id":"1912.05313","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-cognitive-radar-a-revealed","title":"Identifying Cognitive Radars -- Inverse Reinforcement Learning using Revealed Preferences","date":"2019-12-01","arxiv_id":"1912.00331","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-reinforcement-learning-in-2","title":"Near-Optimal Reinforcement Learning in Dynamic Treatment Regimes","date":"2019-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"optimization-for-reinforcement-learning-from","title":"Optimization for Reinforcement Learning: From Single Agent to Cooperative Agents","date":"2019-12-01","arxiv_id":"1912.00498","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-q-learning-with-function-1","title":"Provably Efficient Q-learning with Function Approximation via Distribution Shift Error Checking Oracle","date":"2019-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-bounds-for-learning-state","title":"Regret Bounds for Learning State Representations in Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"text-based-interactive-recommendation-via","title":"Text-Based Interactive Recommendation via Constraint-Augmented Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-importance-weighted-asynchronous-1","title":"IMPACT: Importance Weighted Asynchronous Architectures with Clipped Target Networks","date":"2019-11-30","arxiv_id":"1912.00167","repositories_listed":0,"syntology":null},{"url":null,"slug":"mix-and-match-markov-chains-mixing-times-for","title":"Mix and Match: Markov Chains & Mixing Times for Matching in Rideshare","date":"2019-11-30","arxiv_id":"1912.00225","repositories_listed":0,"syntology":null},{"url":null,"slug":"induction-of-subgoal-automata-for","title":"Induction of Subgoal Automata for Reinforcement Learning","date":"2019-11-29","arxiv_id":"1911.13152","repositories_listed":0,"syntology":null},{"url":null,"slug":"quadratic-q-network-for-learning-continuous","title":"Quadratic Q-network for Learning Continuous Control for Autonomous Vehicles","date":"2019-11-29","arxiv_id":"1912.00074","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithmic-improvements-for-deep","title":"Algorithmic Improvements for Deep Reinforcement Learning applied to Interactive Fiction","date":"2019-11-28","arxiv_id":"1911.12511","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmented-random-search-for-quadcopter","title":"Augmented Random Search for Quadcopter Control: An alternative to Reinforcement Learning","date":"2019-11-28","arxiv_id":"1911.12553","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-neural-relation-extraction-with-1","title":"Improving Neural Relation Extraction with Positive and Unlabeled Learning","date":"2019-11-28","arxiv_id":"1911.12556","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-with-2","title":"Multi-Agent Deep Reinforcement Learning with Adaptive Policies","date":"2019-11-28","arxiv_id":"1912.00949","repositories_listed":0,"syntology":null},{"url":null,"slug":"stigmergic-independent-reinforcement-learning","title":"Stigmergic Independent Reinforcement Learning for Multi-Agent Collaboration","date":"2019-11-28","arxiv_id":"1911.12504","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-adaptive","title":"Adversarial Deep Reinforcement Learning based Adaptive Moving Target Defense","date":"2019-11-27","arxiv_id":"1911.11972","repositories_listed":0,"syntology":null},{"url":null,"slug":"grim-repr-prioritising-generating-important","title":"GRIm-RePR: Prioritising Generating Important Features for Pseudo-Rehearsal","date":"2019-11-27","arxiv_id":"1911.11988","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-fictitious-play-reinforcement","title":"Improving Fictitious Play Reinforcement Learning with Expanding Models","date":"2019-11-27","arxiv_id":"1911.11928","repositories_listed":0,"syntology":null},{"url":null,"slug":"restoring-chaos-using-deep-reinforcement","title":"Restoring Chaos Using Deep Reinforcement Learning","date":"2019-11-27","arxiv_id":"1912.00947","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-tutored-reinforcement-learning-an","title":"Control-Tutored Reinforcement Learning: an application to the Herding Problem","date":"2019-11-26","arxiv_id":"1911.11444","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-portfolio-management-with","title":"A General Framework on Enhancing Portfolio Management with Reinforcement Learning","date":"2019-11-26","arxiv_id":"1911.11880","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-vehicle-mixed-reality-reinforcement","title":"Multi-Vehicle Mixed-Reality Reinforcement Learning for Autonomous Multi-Lane Driving","date":"2019-11-26","arxiv_id":"1911.11699","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-architecture","title":"A Deep Reinforcement Learning Architecture for Multi-stage Optimal Control","date":"2019-11-25","arxiv_id":"1911.10684","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-modulation-and-coding-based-on","title":"Adaptive Modulation and Coding based on Reinforcement Learning for 5G Networks","date":"2019-11-25","arxiv_id":"1912.04030","repositories_listed":0,"syntology":null},{"url":null,"slug":"biologically-inspired-architectures-for","title":"Biologically inspired architectures for sample-efficient deep reinforcement learning","date":"2019-11-25","arxiv_id":"1911.11285","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-multi-driver","title":"Deep Reinforcement Learning for Multi-Driver Vehicle Dispatching and Repositioning Problem","date":"2019-11-25","arxiv_id":"1911.11260","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigate-bias-in-face-recognition-using","title":"Mitigate Bias in Face Recognition using Skewness-Aware Reinforcement Learning","date":"2019-11-25","arxiv_id":"1911.10692","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-a","title":"Multi-Agent Reinforcement Learning: A Selective Overview of Theories and Algorithms","date":"2019-11-24","arxiv_id":"1911.10635","repositories_listed":0,"syntology":null},{"url":null,"slug":"which-channel-to-ask-my-question-personalized","title":"Which Channel to Ask My Question? Personalized Customer Service RequestStream Routing using DeepReinforcement Learning","date":"2019-11-24","arxiv_id":"1911.10521","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-persistent-homology-to-reinforcement","title":"From Persistent Homology to Reinforcement Learning with Applications for Retail Banking","date":"2019-11-23","arxiv_id":"1911.11573","repositories_listed":0,"syntology":null},{"url":null,"slug":"iteratively-refined-interactive-3d-medical","title":"Iteratively-Refined Interactive 3D Medical Image Segmentation with Multi-Agent Reinforcement Learning","date":"2019-11-23","arxiv_id":"1911.10334","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-evolutionary-behavior-in-self","title":"Analysis of Evolutionary Behavior in Self-Learning Media Search Engines","date":"2019-11-22","arxiv_id":"1911.09882","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-trading","title":"Deep Reinforcement Learning for Trading","date":"2019-11-22","arxiv_id":"1911.10107","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-pruning-for-model-compression","title":"Graph Pruning for Model Compression","date":"2019-11-22","arxiv_id":"1911.09817","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-reinforcement-learning-with","title":"Accelerating Reinforcement Learning with Suboptimal Guidance","date":"2019-11-21","arxiv_id":"1911.09391","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-probing-interaction-policies","title":"Agent Probing Interaction Policies","date":"2019-11-21","arxiv_id":"1911.09535","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-drone-mobility-support-using","title":"Efficient Drone Mobility Support Using Reinforcement Learning","date":"2019-11-21","arxiv_id":"1911.09715","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-theoretic-confidence-bounds-for-1","title":"Information-Theoretic Confidence Bounds for Reinforcement Learning","date":"2019-11-21","arxiv_id":"1911.09724","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-alignment-based-imitation-learning-1","title":"State Alignment-based Imitation Learning","date":"2019-11-21","arxiv_id":"1911.10947","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-tale-of-two-timescale-reinforcement","title":"A Tale of Two-Timescale Reinforcement Learning with the Tightest Finite-Time Bound","date":"2019-11-20","arxiv_id":"1911.09157","repositories_listed":0,"syntology":null},{"url":null,"slug":"avoiding-jammers-a-reinforcement-learning","title":"Avoiding Jammers: A Reinforcement Learning Approach","date":"2019-11-20","arxiv_id":"1911.08874","repositories_listed":0,"syntology":null},{"url":null,"slug":"corruption-robust-exploration-in-episodic","title":"Corruption-robust exploration in episodic reinforcement learning","date":"2019-11-20","arxiv_id":"1911.08689","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-cryptocurrency","title":"Deep Reinforcement Learning in Cryptocurrency Market Making","date":"2019-11-20","arxiv_id":"1911.08647","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-policy-learning-robust-to-irreversible","title":"On Policy Learning Robust to Irreversible Events: An Application to Robotic In-Hand Manipulation","date":"2019-11-20","arxiv_id":"1911.08927","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-policies-for-reinforcement-learning-via","title":"Safe Policies for Reinforcement Learning via Primal-Dual Methods","date":"2019-11-20","arxiv_id":"1911.09101","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-online-threat-screening-games-using","title":"Solving Online Threat Screening Games using Constrained Action Space Reinforcement Learning","date":"2019-11-20","arxiv_id":"1911.08799","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-inverse-reinforcement-learning","title":"Decision Making for Autonomous Driving via Augmented Adversarial Inverse Reinforcement Learning","date":"2019-11-19","arxiv_id":"1911.08044","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-privileged-reinforcement-learning-1","title":"Attention-Privileged Reinforcement Learning","date":"2019-11-19","arxiv_id":"1911.08363","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-decorrelation-of-features-using","title":"Efficient decorrelation of features using Gramian in Reinforcement Learning","date":"2019-11-19","arxiv_id":"1911.08610","repositories_listed":0,"syntology":null},{"url":null,"slug":"placement-optimization-of-aerial-base","title":"Placement Optimization of Aerial Base Stations with Deep Reinforcement Learning","date":"2019-11-19","arxiv_id":"1911.08111","repositories_listed":0,"syntology":null},{"url":null,"slug":"variance-reduced-advantage-estimation-with","title":"Variance Reduced Advantage Estimation with $δ$ Hindsight Credit Assignment","date":"2019-11-19","arxiv_id":"1911.08362","repositories_listed":0,"syntology":null},{"url":null,"slug":"comments-on-the-du-kakade-wang-yang-lower","title":"Comments on the Du-Kakade-Wang-Yang Lower Bounds","date":"2019-11-18","arxiv_id":"1911.07910","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-exploration-through-intrinsic","title":"Efficient Exploration through Intrinsic Motivation Learning for Unsupervised Subgoal Discovery in Model-Free Hierarchical Reinforcement Learning","date":"2019-11-18","arxiv_id":"1911.10164","repositories_listed":0,"syntology":null},{"url":null,"slug":"inducing-cooperation-via-team-regret","title":"Inducing Cooperation via Team Regret Minimization based Multi-Agent Deep Reinforcement Learning","date":"2019-11-18","arxiv_id":"1911.07712","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-reinforcement-learning-of","title":"Unsupervised Reinforcement Learning of Transferable Meta-Skills for Embodied Navigation","date":"2019-11-18","arxiv_id":"1911.07450","repositories_listed":0,"syntology":null},{"url":null,"slug":"hebbian-synaptic-modifications-in-spiking","title":"Hebbian Synaptic Modifications in Spiking Neurons that Learn","date":"2019-11-17","arxiv_id":"1911.07247","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-maximum-causal-entropy-for","title":"Generalized Maximum Causal Entropy for Inverse Reinforcement Learning","date":"2019-11-16","arxiv_id":"1911.06928","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-with-missing","title":"Inverse Reinforcement Learning with Missing Data","date":"2019-11-16","arxiv_id":"1911.06930","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-from-imperfect-2","title":"Reinforcement Learning from Imperfect Demonstrations under Soft Expert Guidance","date":"2019-11-16","arxiv_id":"1911.07109","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-co-adaptation-of-morphology","title":"Data-efficient Co-Adaptation of Morphology and Behaviour with Deep Reinforcement Learning","date":"2019-11-15","arxiv_id":"1911.06832","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-exploration-through-latent","title":"Improved Exploration through Latent Trajectory Optimization in Deep Deterministic Policy Gradient","date":"2019-11-15","arxiv_id":"1911.06833","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reduction-from-reinforcement-learning-to-no","title":"A Reduction from Reinforcement Learning to No-Regret Online Learning","date":"2019-11-14","arxiv_id":"1911.05873","repositories_listed":0,"syntology":null},{"url":null,"slug":"atari-fying-the-vehicle-routing-problem-with","title":"Gamifying the Vehicle Routing Problem with Stochastic Requests","date":"2019-11-14","arxiv_id":"1911.05922","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-adaptive","title":"Deep Reinforcement Learning for Adaptive Traffic Signal Control","date":"2019-11-14","arxiv_id":"1911.06294","repositories_listed":0,"syntology":null},{"url":null,"slug":"asymptotics-of-reinforcement-learning-with","title":"Asymptotics of Reinforcement Learning with Neural Networks","date":"2019-11-13","arxiv_id":"1911.07304","repositories_listed":0,"syntology":null},{"url":null,"slug":"buffer-aware-wireless-scheduling-based-on","title":"Buffer-aware Wireless Scheduling based on Deep Reinforcement Learning","date":"2019-11-13","arxiv_id":"1911.05281","repositories_listed":0,"syntology":null},{"url":null,"slug":"kinematic-state-abstraction-and-provably","title":"Kinematic State Abstraction and Provably Efficient Rich-Observation Reinforcement Learning","date":"2019-11-13","arxiv_id":"1911.05815","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-communicate-in-multi-agent","title":"Learning to Communicate in Multi-Agent Reinforcement Learning : A Review","date":"2019-11-13","arxiv_id":"1911.05438","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-driven-test-generation","title":"Reinforcement Learning-Driven Test Generation for Android GUI Applications using Formal Specifications","date":"2019-11-13","arxiv_id":"1911.05403","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-training-in-pommerman-with","title":"Accelerating Training in Pommerman with Imitation and Reinforcement Learning","date":"2019-11-12","arxiv_id":"1911.04947","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-planning-under-partial","title":"Efficient Planning under Partial Observability with Unnormalized Q Functions and Spectral Learning","date":"2019-11-12","arxiv_id":"1911.05010","repositories_listed":0,"syntology":null},{"url":null,"slug":"msdf-a-deep-reinforcement-learning-framework","title":"MSDF: A Deep Reinforcement Learning Framework for Service Function Chain Migration","date":"2019-11-12","arxiv_id":"1911.04801","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-shot-learning-and-behavioral-eligibility","title":"One-shot learning and behavioral eligibility traces in sequential decision making","date":"2019-11-12","arxiv_id":"1707.04192","repositories_listed":0,"syntology":null},{"url":null,"slug":"schedule-earth-observation-satellites-with","title":"Schedule Earth Observation satellites with Deep Reinforcement Learning","date":"2019-11-12","arxiv_id":"1911.05696","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-active-multi-step-reinforcement","title":"Context-aware Active Multi-Step Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04107","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-variational","title":"Reinforcement-Learning-Based Variational Quantum Circuits Optimization for Combinatorial Problems","date":"2019-11-11","arxiv_id":"1911.04574","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-dynamic","title":"Deep Reinforcement Learning Based Dynamic Trajectory Control for UAV-assisted Mobile Edge Computing","date":"2019-11-10","arxiv_id":"1911.03887","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimalistic-attacks-how-little-it-takes-to","title":"Minimalistic Attacks: How Little it Takes to Fool a Deep Reinforcement Learning Policy","date":"2019-11-10","arxiv_id":"1911.03849","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-added-chemical-discovery-using","title":"Value-Added Chemical Discovery Using Reinforcement Learning","date":"2019-11-10","arxiv_id":"1911.07630","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-method","title":"Hierarchical Reinforcement Learning Method for Autonomous Vehicle Behavior Planning","date":"2019-11-09","arxiv_id":"1911.03799","repositories_listed":0,"syntology":null},{"url":null,"slug":"worst-cases-policy-gradients","title":"Worst Cases Policy Gradients","date":"2019-11-09","arxiv_id":"1911.03618","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-bayesian-recurrent-neural-networks-for","title":"Fully Bayesian Recurrent Neural Networks for Safe Reinforcement Learning","date":"2019-11-08","arxiv_id":"1911.03308","repositories_listed":0,"syntology":null},{"url":null,"slug":"option-compatible-reward-inverse","title":"Option Compatible Reward Inverse Reinforcement Learning","date":"2019-11-07","arxiv_id":"1911.02723","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reward-decomposition-for","title":"Distributional Reward Decomposition for Reinforcement Learning","date":"2019-11-06","arxiv_id":"1911.02166","repositories_listed":0,"syntology":null},{"url":null,"slug":"experience-sharing-between-cooperative","title":"Experience Sharing Between Cooperative Reinforcement Learning Agents","date":"2019-11-06","arxiv_id":"1911.02191","repositories_listed":0,"syntology":null},{"url":null,"slug":"mbcal-a-simple-and-efficient-reinforcement","title":"MBCAL: Sample Efficient and Variance Reduced Reinforcement Learning for Recommender Systems","date":"2019-11-06","arxiv_id":"1911.02248","repositories_listed":0,"syntology":null}],"record_sha256":"0390178c9481f0d91bd22d10325cbd68da4e25b525bd34d35aa60bbc5274f81a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}