{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/112","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":112,"pages_in_order":132,"rows_per_page":100,"rows":[11101,11200],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/111","next":"/task/reinforcement-learning/papers/113","papers":[{"url":null,"slug":"automatic-left-atrial-appendage-orifice","title":"Centerline Depth World Reinforcement Learning-based Left Atrial Appendage Orifice Localization","date":"2019-04-02","arxiv_id":"1904.01241","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-and-visualizing-weaknesses-of-deep","title":"Finding and Visualizing Weaknesses of Deep Reinforcement Learning Agents","date":"2019-04-02","arxiv_id":"1904.01318","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-cancer-chemotherapy-schedule-a","title":"Personalized Cancer Chemotherapy Schedule: a numerical comparison of performance and robustness in model-based and model-free scheduling methodologies","date":"2019-04-02","arxiv_id":"1904.01200","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-with-expectation-models","title":"Planning with Expectation Models","date":"2019-04-02","arxiv_id":"1904.01191","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-power-control-for-large-energy","title":"Distributed Power Control for Large Energy Harvesting Networks: A Multi-Agent Deep Reinforcement Learning Approach","date":"2019-04-01","arxiv_id":"1904.00601","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-meta-policy-search","title":"Guided Meta-Policy Search","date":"2019-04-01","arxiv_id":"1904.00956","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-multi-agent-reinforcement-1","title":"Cooperative Multi-Agent Reinforcement Learning Framework for Scalping Trading","date":"2019-03-31","arxiv_id":"1904.00441","repositories_listed":0,"syntology":null},{"url":null,"slug":"power-control-for-wireless-vbr-video","title":"Power Control for Wireless VBR Video Streaming: From Optimization to Reinforcement Learning","date":"2019-03-31","arxiv_id":"1904.00327","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-averse-robust-adversarial-reinforcement","title":"Risk Averse Robust Adversarial Reinforcement Learning","date":"2019-03-31","arxiv_id":"1904.00511","repositories_listed":0,"syntology":null},{"url":null,"slug":"lane-change-decision-making-through-deep","title":"Lane Change Decision-making through Deep Reinforcement Learning with Rule-based Constraints","date":"2019-03-30","arxiv_id":"1904.00231","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-highway-driving-using-deep","title":"Autonomous Highway Driving using Deep Reinforcement Learning","date":"2019-03-29","arxiv_id":"1904.00035","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-reinforcement-learning-with","title":"Improved Reinforcement Learning with Curriculum","date":"2019-03-29","arxiv_id":"1903.12328","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-good-representation-via-continuous","title":"Learning Good Representation via Continuous Attention","date":"2019-03-29","arxiv_id":"1903.12344","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-data-detection-for-mimo-systems-with","title":"Robust Data Detection for MIMO Systems with One-Bit ADCs: A Reinforcement Learning Approach","date":"2019-03-29","arxiv_id":"1903.12546","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-brain-inspired-system-deep-recurrent","title":"Towards Brain-inspired System: Deep Recurrent Reinforcement Learning for Simulated Self-driving Agent","date":"2019-03-29","arxiv_id":"1903.12517","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-surrogate-models-for-sequential","title":"Meta-Learning surrogate models for sequential decision making","date":"2019-03-28","arxiv_id":"1903.11907","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularizing-trajectory-optimization-with","title":"Regularizing Trajectory Optimization with Denoising Autoencoders","date":"2019-03-28","arxiv_id":"1903.11981","repositories_listed":0,"syntology":null},{"url":null,"slug":"wasserstein-dependency-measure-for","title":"Wasserstein Dependency Measure for Representation Learning","date":"2019-03-28","arxiv_id":"1903.11780","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-control-of-stochastic-evolution-a","title":"Dynamic Control of Stochastic Evolution: A Deep Reinforcement Learning Approach to Adaptively Targeting Emergent Drug Resistance","date":"2019-03-27","arxiv_id":"1903.11373","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-storage-management-via-deep-q-networks","title":"Energy Storage Management via Deep Q-Networks","date":"2019-03-26","arxiv_id":"1903.11107","repositories_listed":0,"syntology":null},{"url":null,"slug":"failure-scenario-maker-for-rule-based-agent","title":"Failure-Scenario Maker for Rule-Based Agent using Multi-agent Adversarial Reinforcement Learning and its Application to Autonomous Driving","date":"2019-03-26","arxiv_id":"1903.10654","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-where-to-see-a-novel-attention-model","title":"Learning Where to See: A Novel Attention Model for Automated Immunohistochemical Scoring","date":"2019-03-26","arxiv_id":"1903.10762","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-text-style","title":"Reinforcement Learning Based Text Style Transfer without Parallel Training Corpus","date":"2019-03-26","arxiv_id":"1903.10671","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-use-of-deep-autoencoders-for-efficient","title":"On the use of Deep Autoencoders for Efficient Embedded Reinforcement Learning","date":"2019-03-25","arxiv_id":"1903.10404","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-for-continuous-actions-with-cross","title":"Q-Learning for Continuous Actions with Cross-Entropy Guided Policies","date":"2019-03-25","arxiv_id":"1903.10605","repositories_listed":0,"syntology":null},{"url":null,"slug":"winning-isnt-everything-training-human-like","title":"Winning Isn't Everything: Enhancing Game Development with Intelligent Agents","date":"2019-03-25","arxiv_id":"1903.10545","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-task-discovery-with-limited-supervision-a","title":"Sub-Task Discovery with Limited Supervision: A Constrained Clustering Approach","date":"2019-03-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-logic-guided-safe-reinforcement","title":"Temporal Logic Guided Safe Reinforcement Learning Using Control Barrier Functions","date":"2019-03-23","arxiv_id":"1903.09885","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-hierarchical-reinforcement-learning-1","title":"Deep Hierarchical Reinforcement Learning Based Recommendations via Multi-goals Abstraction","date":"2019-03-22","arxiv_id":"1903.09374","repositories_listed":0,"syntology":null},{"url":null,"slug":"dqn-with-model-based-exploration-efficient","title":"DQN with model-based exploration: efficient learning on environments with sparse rewards","date":"2019-03-22","arxiv_id":"1903.09295","repositories_listed":0,"syntology":null},{"url":null,"slug":"explaining-reinforcement-learning-to-mere","title":"Explaining Reinforcement Learning to Mere Mortals: An Empirical Study","date":"2019-03-22","arxiv_id":"1903.09708","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-safety-in-reinforcement-learning","title":"Improving Safety in Reinforcement Learning Using Model-Based Architectures and Human Intervention","date":"2019-03-22","arxiv_id":"1903.09328","repositories_listed":0,"syntology":null},{"url":null,"slug":"macro-action-reinforcement-learning-with","title":"Macro Action Reinforcement Learning with Sequence Disentanglement using Variational Autoencoder","date":"2019-03-22","arxiv_id":"1903.09366","repositories_listed":0,"syntology":null},{"url":null,"slug":"symbolic-regression-methods-for-reinforcement","title":"Symbolic Regression Methods for Reinforcement Learning","date":"2019-03-22","arxiv_id":"1903.09688","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-off-policy-actor-critic","title":"Distributed off-Policy Actor-Critic Reinforcement Learning with Policy Consensus","date":"2019-03-21","arxiv_id":"1903.09255","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-automatic-construction-of-multi","title":"Towards automatic construction of multi-network models for heterogeneous multi-task learning","date":"2019-03-21","arxiv_id":"1903.09171","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-characterizing-divergence-in-deep-q","title":"Towards Characterizing Divergence in Deep Q-Learning","date":"2019-03-21","arxiv_id":"1903.08894","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmented-memory-networks-for-streaming-based-1","title":"Augmented Memory Networks for Streaming-Based Active One-Shot Learning","date":"2019-03-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcing-classical-planning-for-adversary","title":"Single-step Options for Adversary Driving","date":"2019-03-20","arxiv_id":"1903.08606","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversity-promoting-deep-reinforcement","title":"Diversity-Promoting Deep Reinforcement Learning for Interactive Recommendation","date":"2019-03-19","arxiv_id":"1903.07826","repositories_listed":0,"syntology":null},{"url":null,"slug":"hindsight-generative-adversarial-imitation","title":"Hindsight Generative Adversarial Imitation Learning","date":"2019-03-19","arxiv_id":"1903.07854","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-reciprocity-in-complex-sequential","title":"Learning Reciprocity in Complex Sequential Social Dilemmas","date":"2019-03-19","arxiv_id":"1903.08082","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-prediction-algorithms-and","title":"A Comparison of Prediction Algorithms and Nexting for Short Term Weather Forecasts","date":"2019-03-18","arxiv_id":"1903.07512","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with","title":"Deep Reinforcement Learning with Decorrelation","date":"2019-03-18","arxiv_id":"1903.07765","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-hierarchy-for-learning-and","title":"Exploiting Hierarchy for Learning and Transfer in KL-regularized RL","date":"2019-03-18","arxiv_id":"1903.07438","repositories_listed":0,"syntology":null},{"url":null,"slug":"scheduled-intrinsic-drive-a-hierarchical-take","title":"Scheduled Intrinsic Drive: A Hierarchical Take on Intrinsically Motivated Exploration","date":"2019-03-18","arxiv_id":"1903.07400","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-genomic-evolution-of-neural-network","title":"Adaptive Genomic Evolution of Neural Network Topologies (AGENT) for State-to-Action Mapping in Autonomous Agents","date":"2019-03-17","arxiv_id":"1903.07107","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-proposals-for-sequential-importance","title":"Learning proposals for sequential importance samplers using reinforced variational inference","date":"2019-03-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-query-reformulation-challenges","title":"Multi-agent query reformulation: Challenges and the role of diversity","date":"2019-03-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-for-autonomous","title":"Robust Reinforcement Learning for Autonomous Driving","date":"2019-03-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"online-antenna-tuning-in-heterogeneous","title":"Online Antenna Tuning in Heterogeneous Cellular Networks with Deep Reinforcement Learning","date":"2019-03-15","arxiv_id":"1903.06787","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-distillation-and-value-matching-in","title":"Policy Distillation and Value Matching in Multiagent Reinforcement Learning","date":"2019-03-15","arxiv_id":"1903.06592","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-robot-attract-passersby-without-causing","title":"Can User-Centered Reinforcement Learning Allow a Robot to Attract Passersby without Causing Discomfort?","date":"2019-03-14","arxiv_id":"1903.05881","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-markov-decision-processes-using","title":"No-regret Exploration in Contextual Reinforcement Learning","date":"2019-03-14","arxiv_id":"1903.06187","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-applications-of-bootstrap-in-continuous","title":"On Applications of Bootstrap in Continuous Space Reinforcement Learning","date":"2019-03-14","arxiv_id":"1903.05803","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-reinforcement-learning-based-local","title":"Effective reinforcement learning based local search for the maximum k-plex problem","date":"2019-03-13","arxiv_id":"1903.05537","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-abstraction-for-reinforcement","title":"Resource Abstraction for Reinforcement Learning in Multiagent Congestion Problems","date":"2019-03-13","arxiv_id":"1903.05431","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-design-through-deep","title":"Task-oriented Design through Deep Reinforcement Learning","date":"2019-03-13","arxiv_id":"1903.05271","repositories_listed":0,"syntology":null},{"url":null,"slug":"trajectory-optimization-for-unknown","title":"Trajectory Optimization for Unknown Constrained Systems using Reinforcement Learning","date":"2019-03-13","arxiv_id":"1903.05751","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-reinforcement-learning-for","title":"A Review of Reinforcement Learning for Autonomous Building Energy Management","date":"2019-03-12","arxiv_id":"1903.05196","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-multi-agent-reinforcement-learning-with-1","title":"Deep Multi-Agent Reinforcement Learning with Discrete-Continuous Hybrid Action Spaces","date":"2019-03-12","arxiv_id":"1903.04959","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-minibatch-stochastic-gradient-1","title":"Accelerating Minibatch Stochastic Gradient Descent using Typicality Sampling","date":"2019-03-11","arxiv_id":"1903.04192","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-for-molecular-generation-and","title":"Deep learning for molecular design - a review of the state of the art","date":"2019-03-11","arxiv_id":"1903.04388","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-of-volume-guided","title":"Deep Reinforcement Learning of Volume-guided Progressive View Inpainting for 3D Point Scene Completion from a Single Depth Image","date":"2019-03-10","arxiv_id":"1903.04019","repositories_listed":0,"syntology":null},{"url":null,"slug":"deeppool-distributed-model-free-algorithm-for","title":"DeepPool: Distributed Model-free Algorithm for Ride-sharing using Deep Reinforcement Learning","date":"2019-03-09","arxiv_id":"1903.03882","repositories_listed":0,"syntology":null},{"url":null,"slug":"orthogonal-estimation-of-wasserstein","title":"Orthogonal Estimation of Wasserstein Distances","date":"2019-03-09","arxiv_id":"1903.03784","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-memory-transformer-for-embodied-agents","title":"Scene Memory Transformer for Embodied Agents in Long-Horizon Tasks","date":"2019-03-09","arxiv_id":"1903.03878","repositories_listed":0,"syntology":null},{"url":null,"slug":"successive-over-relaxation-q-learning","title":"Successive Over Relaxation Q-Learning","date":"2019-03-09","arxiv_id":"1903.03812","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cooperative-game-for-automated-learning-of","title":"A cooperative game for automated learning of elasto-plasticity knowledge graphs and models with AI-guided experimentation","date":"2019-03-08","arxiv_id":"1903.04307","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-robustness-and-safety-for-autonomous","title":"Improved Robustness and Safety for Autonomous Vehicle Control with Adversarial Reinforcement Learning","date":"2019-03-08","arxiv_id":"1903.03642","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-skin-condition-classification-with-1","title":"Improving Skin Condition Classification with a Visual Symptom Checker Trained using Reinforcement Learning","date":"2019-03-08","arxiv_id":"1903.03495","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-self-game-play-agents-for","title":"Learning Self-Game-Play Agents for Combinatorial Optimization Problems","date":"2019-03-08","arxiv_id":"1903.03674","repositories_listed":0,"syntology":null},{"url":null,"slug":"pixel-attentive-policy-gradient-for-multi","title":"Pixel-Attentive Policy Gradient for Multi-Fingered Grasping in Cluttered Scenes","date":"2019-03-08","arxiv_id":"1903.03227","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-sophisticated-dispatching-strategy","title":"Can Sophisticated Dispatching Strategy Acquired by Reinforcement Learning? - A Case Study in Dynamic Courier Dispatching System","date":"2019-03-07","arxiv_id":"1903.02716","repositories_listed":0,"syntology":null},{"url":null,"slug":"rloc-neurobiologically-inspired-hierarchical","title":"RLOC: Neurobiologically Inspired Hierarchical Reinforcement Learning Algorithm for Continuous Control of Nonlinear Dynamical Systems","date":"2019-03-07","arxiv_id":"1903.03064","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-random-search-is-not-enough-sample","title":"Provably Robust Blackbox Optimization for Reinforcement Learning","date":"2019-03-07","arxiv_id":"1903.02993","repositories_listed":0,"syntology":null},{"url":null,"slug":"minigo-a-case-study-in-reproducing","title":"Minigo: A Case Study in Reproducing Reinforcement Learning Research","date":"2019-03-06","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-guided-deep-reinforcement-learning-via","title":"Safety-Guided Deep Reinforcement Learning via Online Gaussian Process Estimation","date":"2019-03-06","arxiv_id":"1903.02526","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-chemical-plant-operation","title":"Synthesizing Chemical Plant Operation Procedures using Knowledge, Dynamic Simulation and Deep Reinforcement Learning","date":"2019-03-06","arxiv_id":"1903.02183","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-in-task-space-to-speed-up-and-guide","title":"Training in Task Space to Speed Up and Guide Reinforcement Learning","date":"2019-03-06","arxiv_id":"1903.02219","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-artificial-intelligence","title":"Understanding the Artificial Intelligence Clinician and optimal treatment strategies for sepsis in intensive care","date":"2019-03-06","arxiv_id":"1903.02345","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-world-models-for-pseudo-rehearsal-in","title":"Continual Learning Using World Models for Pseudo-Rehearsal","date":"2019-03-06","arxiv_id":"1903.02647","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dynamics-model-in-reinforcement","title":"Learning Dynamics Model in Reinforcement Learning by Incorporating the Long Term Future","date":"2019-03-05","arxiv_id":"1903.01599","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-data-poisoning-attack","title":"Online Data Poisoning Attack","date":"2019-03-05","arxiv_id":"1903.01666","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-chinese-checkers-with","title":"Towards Understanding Chinese Checkers with Heuristics, Monte Carlo Tree Search, and Deep Reinforcement Learning","date":"2019-03-05","arxiv_id":"1903.01747","repositories_listed":0,"syntology":null},{"url":null,"slug":"microscopic-traffic-simulation-by-cooperative","title":"Microscopic Traffic Simulation by Cooperative Multi-agent Deep Reinforcement Learning","date":"2019-03-04","arxiv_id":"1903.01365","repositories_listed":0,"syntology":null},{"url":null,"slug":"strong-asymptotic-optimality-in-general","title":"A Strongly Asymptotically Optimal Agent in General Environments","date":"2019-03-04","arxiv_id":"1903.01021","repositories_listed":0,"syntology":null},{"url":null,"slug":"hacking-google-recaptcha-v3-using","title":"Hacking Google reCAPTCHA v3 using Reinforcement Learning","date":"2019-03-03","arxiv_id":"1903.01003","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-framework-for-regularized","title":"A Regularized Approach to Sparse Optimal Policy in Reinforcement Learning","date":"2019-03-02","arxiv_id":"1903.00725","repositories_listed":0,"syntology":null},{"url":null,"slug":"automating-predictive-modeling-process-using","title":"Automating Predictive Modeling Process using Reinforcement Learning","date":"2019-03-02","arxiv_id":"1903.00743","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-options-for-exploration-by","title":"Discovering Options for Exploration by Minimizing Cover Time","date":"2019-03-02","arxiv_id":"1903.00606","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnidrl-robust-pedestrian-detection-using","title":"OmniDRL: Robust Pedestrian Detection using Deep Reinforcement Learning on Omnidirectional Cameras","date":"2019-03-02","arxiv_id":"1903.00676","repositories_listed":0,"syntology":null},{"url":null,"slug":"straight-to-the-point-reinforcement-learning","title":"Straight to the point: reinforcement learning for user guidance in ultrasound","date":"2019-03-02","arxiv_id":"1903.00586","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-curriculum","title":"Reinforcement Learning based Curriculum Optimization for Neural Machine Translation","date":"2019-02-28","arxiv_id":"1903.00041","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-caching-via-deep-reinforcement","title":"Deep Reinforcement Learning for Adaptive Caching in Hierarchical Content Delivery Networks","date":"2019-02-27","arxiv_id":"1902.10301","repositories_listed":0,"syntology":null},{"url":null,"slug":"atomistic-structure-learning","title":"Atomistic structure learning","date":"2019-02-27","arxiv_id":"1902.10501","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-edge-caching-via-reinforcement","title":"Distributed Edge Caching via Reinforcement Learning in Fog Radio Access Networks","date":"2019-02-27","arxiv_id":"1902.10574","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-constrained-advertising-keyword","title":"Domain-Constrained Advertising Keyword Generation","date":"2019-02-27","arxiv_id":"1902.10374","repositories_listed":0,"syntology":null},{"url":null,"slug":"introspection-learning","title":"Introspection Learning","date":"2019-02-27","arxiv_id":"1902.10754","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-packet-classification","title":"Neural Packet Classification","date":"2019-02-27","arxiv_id":"1902.10319","repositories_listed":0,"syntology":null}],"record_sha256":"b77194f09ccd9233f58753cb5f25c1f292b65c20c1f70a13675acd6b4754bb1c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}