{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/136","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":136,"pages_in_order":152,"rows_per_page":100,"rows":[13501,13600],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/135","next":"/task/reinforcement-learning-1/papers/137","papers":[{"url":null,"slug":"a-short-survey-on-memory-based-reinforcement","title":"A Short Survey On Memory Based Reinforcement Learning","date":"2019-04-14","arxiv_id":"1904.06736","repositories_listed":0,"syntology":null},{"url":null,"slug":"dot-to-dot-achieving-structured-robotic","title":"Dot-to-Dot: Explainable Hierarchical Reinforcement Learning for Robotic Manipulation","date":"2019-04-14","arxiv_id":"1904.06703","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-scheduling-function-design-in-sdn","title":"Effective Scheduling Function Design in SDN through Deep Reinforcement Learning","date":"2019-04-12","arxiv_id":"1904.06039","repositories_listed":0,"syntology":null},{"url":null,"slug":"similarities-between-policy-gradient-methods","title":"Similarities between policy gradient methods (PGM) in Reinforcement learning (RL) and supervised learning (SL)","date":"2019-04-12","arxiv_id":"1904.06260","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-flow-improve-upon-your-teachers-1","title":"Knowledge Flow: Improve Upon Your Teachers","date":"2019-04-11","arxiv_id":"1904.05878","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-reinforcement-learning-for","title":"Model-Free Reinforcement Learning for Financial Portfolios: A Brief Survey","date":"2019-04-10","arxiv_id":"1904.04973","repositories_listed":0,"syntology":null},{"url":null,"slug":"safer-deep-rl-with-shallow-mcts-a-case-study","title":"Safer Deep RL with Shallow MCTS: A Case Study in Pommerman","date":"2019-04-10","arxiv_id":"1904.05759","repositories_listed":0,"syntology":null},{"url":null,"slug":"190407961","title":"RL-Based User Association and Resource Allocation for Multi-UAV enabled MEC","date":"2019-04-08","arxiv_id":"1904.07961","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-pro-level-ai-for-real-time-fighting","title":"Creating Pro-Level AI for a Real-Time Fighting Game Using Deep Reinforcement Learning","date":"2019-04-08","arxiv_id":"1904.03821","repositories_listed":0,"syntology":null},{"url":null,"slug":"jam-me-if-you-can-defeating-jammer-with-deep","title":"\"Jam Me If You Can'': Defeating Jammer with Deep Dueling Neural Network Architecture and Ambient Backscattering Augmented Communications","date":"2019-04-08","arxiv_id":"1904.03897","repositories_listed":0,"syntology":null},{"url":null,"slug":"teaching-gans-to-sketch-in-vector-format","title":"Teaching GANs to Sketch in Vector Format","date":"2019-04-07","arxiv_id":"1904.03620","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-imitation-in-heterogeneous-action","title":"Reinforced Imitation in Heterogeneous Action Space","date":"2019-04-06","arxiv_id":"1904.03438","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-attention-that","title":"Reinforcement Learning with Attention that Works: A Self-Supervised Approach","date":"2019-04-06","arxiv_id":"1904.03367","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-preference-actor-critic","title":"Multi-Preference Actor Critic","date":"2019-04-05","arxiv_id":"1904.03295","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-adapting-goals-allow-transfer-of","title":"Self-Adapting Goals Allow Transfer of Predictive Models to New Tasks","date":"2019-04-04","arxiv_id":"1904.02435","repositories_listed":0,"syntology":null},{"url":null,"slug":"paintbot-a-reinforcement-learning-approach","title":"PaintBot: A Reinforcement Learning Approach for Natural Media Painting","date":"2019-04-03","arxiv_id":"1904.02201","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-left-atrial-appendage-orifice","title":"Centerline Depth World Reinforcement Learning-based Left Atrial Appendage Orifice Localization","date":"2019-04-02","arxiv_id":"1904.01241","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-and-visualizing-weaknesses-of-deep","title":"Finding and Visualizing Weaknesses of Deep Reinforcement Learning Agents","date":"2019-04-02","arxiv_id":"1904.01318","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-cancer-chemotherapy-schedule-a","title":"Personalized Cancer Chemotherapy Schedule: a numerical comparison of performance and robustness in model-based and model-free scheduling methodologies","date":"2019-04-02","arxiv_id":"1904.01200","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-power-control-for-large-energy","title":"Distributed Power Control for Large Energy Harvesting Networks: A Multi-Agent Deep Reinforcement Learning Approach","date":"2019-04-01","arxiv_id":"1904.00601","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-meta-policy-search","title":"Guided Meta-Policy Search","date":"2019-04-01","arxiv_id":"1904.00956","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-multi-agent-reinforcement-1","title":"Cooperative Multi-Agent Reinforcement Learning Framework for Scalping Trading","date":"2019-03-31","arxiv_id":"1904.00441","repositories_listed":0,"syntology":null},{"url":null,"slug":"power-control-for-wireless-vbr-video","title":"Power Control for Wireless VBR Video Streaming: From Optimization to Reinforcement Learning","date":"2019-03-31","arxiv_id":"1904.00327","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-averse-robust-adversarial-reinforcement","title":"Risk Averse Robust Adversarial Reinforcement Learning","date":"2019-03-31","arxiv_id":"1904.00511","repositories_listed":0,"syntology":null},{"url":null,"slug":"lane-change-decision-making-through-deep","title":"Lane Change Decision-making through Deep Reinforcement Learning with Rule-based Constraints","date":"2019-03-30","arxiv_id":"1904.00231","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-highway-driving-using-deep","title":"Autonomous Highway Driving using Deep Reinforcement Learning","date":"2019-03-29","arxiv_id":"1904.00035","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-reinforcement-learning-with","title":"Improved Reinforcement Learning with Curriculum","date":"2019-03-29","arxiv_id":"1903.12328","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-good-representation-via-continuous","title":"Learning Good Representation via Continuous Attention","date":"2019-03-29","arxiv_id":"1903.12344","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-data-detection-for-mimo-systems-with","title":"Robust Data Detection for MIMO Systems with One-Bit ADCs: A Reinforcement Learning Approach","date":"2019-03-29","arxiv_id":"1903.12546","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-brain-inspired-system-deep-recurrent","title":"Towards Brain-inspired System: Deep Recurrent Reinforcement Learning for Simulated Self-driving Agent","date":"2019-03-29","arxiv_id":"1903.12517","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-surrogate-models-for-sequential","title":"Meta-Learning surrogate models for sequential decision making","date":"2019-03-28","arxiv_id":"1903.11907","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularizing-trajectory-optimization-with","title":"Regularizing Trajectory Optimization with Denoising Autoencoders","date":"2019-03-28","arxiv_id":"1903.11981","repositories_listed":0,"syntology":null},{"url":null,"slug":"wasserstein-dependency-measure-for","title":"Wasserstein Dependency Measure for Representation Learning","date":"2019-03-28","arxiv_id":"1903.11780","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-relation-between-maximum","title":"Understanding the Relation Between Maximum-Entropy Inverse Reinforcement Learning and Behaviour Cloning","date":"2019-03-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-storage-management-via-deep-q-networks","title":"Energy Storage Management via Deep Q-Networks","date":"2019-03-26","arxiv_id":"1903.11107","repositories_listed":0,"syntology":null},{"url":null,"slug":"failure-scenario-maker-for-rule-based-agent","title":"Failure-Scenario Maker for Rule-Based Agent using Multi-agent Adversarial Reinforcement Learning and its Application to Autonomous Driving","date":"2019-03-26","arxiv_id":"1903.10654","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-text-style","title":"Reinforcement Learning Based Text Style Transfer without Parallel Training Corpus","date":"2019-03-26","arxiv_id":"1903.10671","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactions-between-representation-learning","title":"Interactions between Representation Learning and Supervision","date":"2019-03-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-use-of-deep-autoencoders-for-efficient","title":"On the use of Deep Autoencoders for Efficient Embedded Reinforcement Learning","date":"2019-03-25","arxiv_id":"1903.10404","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-for-continuous-actions-with-cross","title":"Q-Learning for Continuous Actions with Cross-Entropy Guided Policies","date":"2019-03-25","arxiv_id":"1903.10605","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-task-discovery-with-limited-supervision-a","title":"Sub-Task Discovery with Limited Supervision: A Constrained Clustering Approach","date":"2019-03-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-program-planner-for-structured","title":"Neural Program Planner for Structured Predictions","date":"2019-03-23","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-logic-guided-safe-reinforcement","title":"Temporal Logic Guided Safe Reinforcement Learning Using Control Barrier Functions","date":"2019-03-23","arxiv_id":"1903.09885","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-hierarchical-reinforcement-learning-1","title":"Deep Hierarchical Reinforcement Learning Based Recommendations via Multi-goals Abstraction","date":"2019-03-22","arxiv_id":"1903.09374","repositories_listed":0,"syntology":null},{"url":null,"slug":"dqn-with-model-based-exploration-efficient","title":"DQN with model-based exploration: efficient learning on environments with sparse rewards","date":"2019-03-22","arxiv_id":"1903.09295","repositories_listed":0,"syntology":null},{"url":null,"slug":"explaining-reinforcement-learning-to-mere","title":"Explaining Reinforcement Learning to Mere Mortals: An Empirical Study","date":"2019-03-22","arxiv_id":"1903.09708","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-safety-in-reinforcement-learning","title":"Improving Safety in Reinforcement Learning Using Model-Based Architectures and Human Intervention","date":"2019-03-22","arxiv_id":"1903.09328","repositories_listed":0,"syntology":null},{"url":null,"slug":"macro-action-reinforcement-learning-with","title":"Macro Action Reinforcement Learning with Sequence Disentanglement using Variational Autoencoder","date":"2019-03-22","arxiv_id":"1903.09366","repositories_listed":0,"syntology":null},{"url":null,"slug":"symbolic-regression-methods-for-reinforcement","title":"Symbolic Regression Methods for Reinforcement Learning","date":"2019-03-22","arxiv_id":"1903.09688","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-off-policy-actor-critic","title":"Distributed off-Policy Actor-Critic Reinforcement Learning with Policy Consensus","date":"2019-03-21","arxiv_id":"1903.09255","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmented-memory-networks-for-streaming-based-1","title":"Augmented Memory Networks for Streaming-Based Active One-Shot Learning","date":"2019-03-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcing-classical-planning-for-adversary","title":"Single-step Options for Adversary Driving","date":"2019-03-20","arxiv_id":"1903.08606","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversity-promoting-deep-reinforcement","title":"Diversity-Promoting Deep Reinforcement Learning for Interactive Recommendation","date":"2019-03-19","arxiv_id":"1903.07826","repositories_listed":0,"syntology":null},{"url":null,"slug":"hindsight-generative-adversarial-imitation","title":"Hindsight Generative Adversarial Imitation Learning","date":"2019-03-19","arxiv_id":"1903.07854","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-prediction-algorithms-and","title":"A Comparison of Prediction Algorithms and Nexting for Short Term Weather Forecasts","date":"2019-03-18","arxiv_id":"1903.07512","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with","title":"Deep Reinforcement Learning with Decorrelation","date":"2019-03-18","arxiv_id":"1903.07765","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-hierarchy-for-learning-and","title":"Exploiting Hierarchy for Learning and Transfer in KL-regularized RL","date":"2019-03-18","arxiv_id":"1903.07438","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-genomic-evolution-of-neural-network","title":"Adaptive Genomic Evolution of Neural Network Topologies (AGENT) for State-to-Action Mapping in Autonomous Agents","date":"2019-03-17","arxiv_id":"1903.07107","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-proposals-for-sequential-importance","title":"Learning proposals for sequential importance samplers using reinforced variational inference","date":"2019-03-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-query-reformulation-challenges","title":"Multi-agent query reformulation: Challenges and the role of diversity","date":"2019-03-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-for-autonomous","title":"Robust Reinforcement Learning for Autonomous Driving","date":"2019-03-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"online-antenna-tuning-in-heterogeneous","title":"Online Antenna Tuning in Heterogeneous Cellular Networks with Deep Reinforcement Learning","date":"2019-03-15","arxiv_id":"1903.06787","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-distillation-and-value-matching-in","title":"Policy Distillation and Value Matching in Multiagent Reinforcement Learning","date":"2019-03-15","arxiv_id":"1903.06592","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-robot-attract-passersby-without-causing","title":"Can User-Centered Reinforcement Learning Allow a Robot to Attract Passersby without Causing Discomfort?","date":"2019-03-14","arxiv_id":"1903.05881","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-markov-decision-processes-using","title":"No-regret Exploration in Contextual Reinforcement Learning","date":"2019-03-14","arxiv_id":"1903.06187","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-applications-of-bootstrap-in-continuous","title":"On Applications of Bootstrap in Continuous Space Reinforcement Learning","date":"2019-03-14","arxiv_id":"1903.05803","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-reinforcement-learning-based-local","title":"Effective reinforcement learning based local search for the maximum k-plex problem","date":"2019-03-13","arxiv_id":"1903.05537","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-abstraction-for-reinforcement","title":"Resource Abstraction for Reinforcement Learning in Multiagent Congestion Problems","date":"2019-03-13","arxiv_id":"1903.05431","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-design-through-deep","title":"Task-oriented Design through Deep Reinforcement Learning","date":"2019-03-13","arxiv_id":"1903.05271","repositories_listed":0,"syntology":null},{"url":null,"slug":"trajectory-optimization-for-unknown","title":"Trajectory Optimization for Unknown Constrained Systems using Reinforcement Learning","date":"2019-03-13","arxiv_id":"1903.05751","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-reinforcement-learning-for","title":"A Review of Reinforcement Learning for Autonomous Building Energy Management","date":"2019-03-12","arxiv_id":"1903.05196","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-multi-agent-reinforcement-learning-with-1","title":"Deep Multi-Agent Reinforcement Learning with Discrete-Continuous Hybrid Action Spaces","date":"2019-03-12","arxiv_id":"1903.04959","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-minibatch-stochastic-gradient-1","title":"Accelerating Minibatch Stochastic Gradient Descent using Typicality Sampling","date":"2019-03-11","arxiv_id":"1903.04192","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-for-molecular-generation-and","title":"Deep learning for molecular design - a review of the state of the art","date":"2019-03-11","arxiv_id":"1903.04388","repositories_listed":0,"syntology":null},{"url":null,"slug":"deeppool-distributed-model-free-algorithm-for","title":"DeepPool: Distributed Model-free Algorithm for Ride-sharing using Deep Reinforcement Learning","date":"2019-03-09","arxiv_id":"1903.03882","repositories_listed":0,"syntology":null},{"url":null,"slug":"orthogonal-estimation-of-wasserstein","title":"Orthogonal Estimation of Wasserstein Distances","date":"2019-03-09","arxiv_id":"1903.03784","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-memory-transformer-for-embodied-agents","title":"Scene Memory Transformer for Embodied Agents in Long-Horizon Tasks","date":"2019-03-09","arxiv_id":"1903.03878","repositories_listed":0,"syntology":null},{"url":null,"slug":"successive-over-relaxation-q-learning","title":"Successive Over Relaxation Q-Learning","date":"2019-03-09","arxiv_id":"1903.03812","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cooperative-game-for-automated-learning-of","title":"A cooperative game for automated learning of elasto-plasticity knowledge graphs and models with AI-guided experimentation","date":"2019-03-08","arxiv_id":"1903.04307","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-robustness-and-safety-for-autonomous","title":"Improved Robustness and Safety for Autonomous Vehicle Control with Adversarial Reinforcement Learning","date":"2019-03-08","arxiv_id":"1903.03642","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-skin-condition-classification-with-1","title":"Improving Skin Condition Classification with a Visual Symptom Checker Trained using Reinforcement Learning","date":"2019-03-08","arxiv_id":"1903.03495","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-self-game-play-agents-for","title":"Learning Self-Game-Play Agents for Combinatorial Optimization Problems","date":"2019-03-08","arxiv_id":"1903.03674","repositories_listed":0,"syntology":null},{"url":null,"slug":"pixel-attentive-policy-gradient-for-multi","title":"Pixel-Attentive Policy Gradient for Multi-Fingered Grasping in Cluttered Scenes","date":"2019-03-08","arxiv_id":"1903.03227","repositories_listed":0,"syntology":null},{"url":null,"slug":"rloc-neurobiologically-inspired-hierarchical","title":"RLOC: Neurobiologically Inspired Hierarchical Reinforcement Learning Algorithm for Continuous Control of Nonlinear Dynamical Systems","date":"2019-03-07","arxiv_id":"1903.03064","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-random-search-is-not-enough-sample","title":"Provably Robust Blackbox Optimization for Reinforcement Learning","date":"2019-03-07","arxiv_id":"1903.02993","repositories_listed":0,"syntology":null},{"url":null,"slug":"minigo-a-case-study-in-reproducing","title":"Minigo: A Case Study in Reproducing Reinforcement Learning Research","date":"2019-03-06","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-guided-deep-reinforcement-learning-via","title":"Safety-Guided Deep Reinforcement Learning via Online Gaussian Process Estimation","date":"2019-03-06","arxiv_id":"1903.02526","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-chemical-plant-operation","title":"Synthesizing Chemical Plant Operation Procedures using Knowledge, Dynamic Simulation and Deep Reinforcement Learning","date":"2019-03-06","arxiv_id":"1903.02183","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-in-task-space-to-speed-up-and-guide","title":"Training in Task Space to Speed Up and Guide Reinforcement Learning","date":"2019-03-06","arxiv_id":"1903.02219","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-world-models-for-pseudo-rehearsal-in","title":"Continual Learning Using World Models for Pseudo-Rehearsal","date":"2019-03-06","arxiv_id":"1903.02647","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dynamics-model-in-reinforcement","title":"Learning Dynamics Model in Reinforcement Learning by Incorporating the Long Term Future","date":"2019-03-05","arxiv_id":"1903.01599","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-data-poisoning-attack","title":"Online Data Poisoning Attack","date":"2019-03-05","arxiv_id":"1903.01666","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-chinese-checkers-with","title":"Towards Understanding Chinese Checkers with Heuristics, Monte Carlo Tree Search, and Deep Reinforcement Learning","date":"2019-03-05","arxiv_id":"1903.01747","repositories_listed":0,"syntology":null},{"url":null,"slug":"microscopic-traffic-simulation-by-cooperative","title":"Microscopic Traffic Simulation by Cooperative Multi-agent Deep Reinforcement Learning","date":"2019-03-04","arxiv_id":"1903.01365","repositories_listed":0,"syntology":null},{"url":null,"slug":"hacking-google-recaptcha-v3-using","title":"Hacking Google reCAPTCHA v3 using Reinforcement Learning","date":"2019-03-03","arxiv_id":"1903.01003","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-framework-for-regularized","title":"A Regularized Approach to Sparse Optimal Policy in Reinforcement Learning","date":"2019-03-02","arxiv_id":"1903.00725","repositories_listed":0,"syntology":null},{"url":null,"slug":"automating-predictive-modeling-process-using","title":"Automating Predictive Modeling Process using Reinforcement Learning","date":"2019-03-02","arxiv_id":"1903.00743","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-options-for-exploration-by","title":"Discovering Options for Exploration by Minimizing Cover Time","date":"2019-03-02","arxiv_id":"1903.00606","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnidrl-robust-pedestrian-detection-using","title":"OmniDRL: Robust Pedestrian Detection using Deep Reinforcement Learning on Omnidirectional Cameras","date":"2019-03-02","arxiv_id":"1903.00676","repositories_listed":0,"syntology":null},{"url":null,"slug":"straight-to-the-point-reinforcement-learning","title":"Straight to the point: reinforcement learning for user guidance in ultrasound","date":"2019-03-02","arxiv_id":"1903.00586","repositories_listed":0,"syntology":null}],"record_sha256":"6a01deb461ff701e5bfa17fe305e923aef610b7df6d4fe30b6eba28872dcefb7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}