{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/15","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":15,"pages_in_order":18,"rows_per_page":100,"rows":[1401,1500],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning/papers/14","next":"/method/q-learning/papers/16","papers":[{"paper":null,"slug":"multi-pseudo-q-learning-based-deterministic","title":"Multi Pseudo Q-learning Based Deterministic Policy Gradient for Tracking Control of Autonomous Underwater Vehicles","date":"2019-09-07","arxiv_id":"1909.03204","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-reinforcement-learning-based-approach-for","title":"Reinforcement Learning for Joint Optimization of Multiple Rewards","date":"2019-09-06","arxiv_id":"1909.02940","n_code_links":0,"syntology":null},{"paper":null,"slug":"encoders-and-decoders-for-quantum-expander","title":"Encoders and Decoders for Quantum Expander Codes Using Machine Learning","date":"2019-09-06","arxiv_id":"1909.02945","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-data-enhanced-traffic-flow-monitoring-in","title":"Q-DATA: Enhanced Traffic Flow Monitoring in Software-Defined Networks applying Q-learning","date":"2019-09-04","arxiv_id":"1909.01544","n_code_links":0,"syntology":null},{"paper":null,"slug":"solving-discounted-stochastic-two-player","title":"Solving Discounted Stochastic Two-Player Games with Near-Optimal Time and Sample Complexity","date":"2019-08-29","arxiv_id":"1908.11071","n_code_links":0,"syntology":null},{"paper":"/paper/intelligent-active-queue-management-using","slug":"intelligent-active-queue-management-using","title":"Intelligent Active Queue Management Using Explicit Congestion Notification","date":"2019-08-28","arxiv_id":"1909.08386","n_code_links":1,"syntology":null},{"paper":null,"slug":"networked-control-of-nonlinear-systems-under","title":"Networked Control of Nonlinear Systems under Partial Observation Using Continuous Deep Q-Learning","date":"2019-08-28","arxiv_id":"1908.10722","n_code_links":0,"syntology":null},{"paper":null,"slug":"stmarl-a-spatio-temporal-multi-agent","title":"STMARL: A Spatio-Temporal Multi-Agent Reinforcement Learning Approach for Cooperative Traffic Light Control","date":"2019-08-28","arxiv_id":"1908.10577","n_code_links":0,"syntology":null},{"paper":"/paper/performing-deep-recurrent-double-q-learning","slug":"performing-deep-recurrent-double-q-learning","title":"Performing Deep Recurrent Double Q-Learning for Atari Games","date":"2019-08-16","arxiv_id":"1908.06040","n_code_links":2,"syntology":null},{"paper":"/paper/learn-how-to-cook-a-new-recipe-in-a-new-house","slug":"learn-how-to-cook-a-new-recipe-in-a-new-house","title":"Learn How to Cook a New Recipe in a New House: Using Map Familiarization, Curriculum Learning, and Bandit Feedback to Learn Families of Text-Based Adventure Games","date":"2019-08-13","arxiv_id":"1908.04777","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yinxusen/deepword"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-scale-traffic-signal-control-using-a","title":"Large-Scale Traffic Signal Control Using a Novel Multi-Agent Reinforcement Learning","date":"2019-08-10","arxiv_id":"1908.03761","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-mind-defeating-stealthy-dos-attacks-in-sdn","title":"Q-MIND: Defeating Stealthy DoS Attacks in SDN with a Machine-learning based Defense Framework","date":"2019-07-27","arxiv_id":"1907.11887","n_code_links":0,"syntology":null},{"paper":"/paper/striving-for-simplicity-in-off-policy-deep","slug":"striving-for-simplicity-in-off-policy-deep","title":"An Optimistic Perspective on Offline Reinforcement Learning","date":"2019-07-10","arxiv_id":"1907.04543","n_code_links":1,"syntology":null},{"paper":"/paper/way-off-policy-batch-deep-reinforcement","slug":"way-off-policy-batch-deep-reinforcement","title":"Way Off-Policy Batch Deep Reinforcement Learning of Implicit Human Preferences in Dialog","date":"2019-06-30","arxiv_id":"1907.00456","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["natashamjaques/neural_chat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"q-learning-inspired-self-tuning-for-energy","title":"Q-Learning Inspired Self-Tuning for Energy Efficiency in HPC","date":"2019-06-26","arxiv_id":"1906.10970","n_code_links":0,"syntology":null},{"paper":"/paper/towards-empathic-deep-q-learning","slug":"towards-empathic-deep-q-learning","title":"Towards Empathic Deep Q-Learning","date":"2019-06-26","arxiv_id":"1906.10918","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bartbussmann/EmpathicDQN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-causal-state-representations-of","title":"Learning Causal State Representations of Partially Observable Environments","date":"2019-06-25","arxiv_id":"1906.10437","n_code_links":0,"syntology":null},{"paper":null,"slug":"in-hindsight-a-smooth-reward-for-steady","title":"In Hindsight: A Smooth Reward for Steady Exploration","date":"2019-06-24","arxiv_id":"1906.09781","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimal-use-of-experience-in-first-person","title":"Optimal Use of Experience in First Person Shooter Environments","date":"2019-06-24","arxiv_id":"1906.09734","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-networks-with-motivation","title":"Neural networks with motivation","date":"2019-06-23","arxiv_id":"1906.09528","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-trajectory","title":"Reinforcement Learning-Based Trajectory Design for the Aerial Base Stations","date":"2019-06-23","arxiv_id":"1906.09550","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-models-of-human","slug":"reinforcement-learning-models-of-human","title":"A Story of Two Streams: Reinforcement Learning Models from Human Behavior and Neuropsychiatry","date":"2019-06-21","arxiv_id":"1906.11286","n_code_links":1,"syntology":null},{"paper":"/paper/split-q-learning-reinforcement-learning-with","slug":"split-q-learning-reinforcement-learning-with","title":"Split Q Learning: Reinforcement Learning with Two-Stream Rewards","date":"2019-06-21","arxiv_id":"1906.12350","n_code_links":1,"syntology":null},{"paper":null,"slug":"solution-of-two-player-zero-sum-game-by","title":"A Generalized Minimax Q-learning Algorithm for Two-Player Zero-Sum Stochastic Games","date":"2019-06-16","arxiv_id":"1906.06659","n_code_links":0,"syntology":null},{"paper":"/paper/boosting-soft-actor-critic-emphasizing-recent","slug":"boosting-soft-actor-critic-emphasizing-recent","title":"Boosting Soft Actor-Critic: Emphasizing Recent Experience without Forgetting the Past","date":"2019-06-10","arxiv_id":"1906.04009","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"escaping-the-state-of-nature-a-hobbesian","title":"Escaping the State of Nature: A Hobbesian Approach to Cooperation in Multi-agent Reinforcement Learning","date":"2019-06-05","arxiv_id":"1906.09874","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploration-with-unreliable-intrinsic-reward","title":"Exploration with Unreliable Intrinsic Reward in Multi-Agent Reinforcement Learning","date":"2019-06-05","arxiv_id":"1906.02138","n_code_links":0,"syntology":null},{"paper":null,"slug":"risk-sensitive-compact-decision-trees-for","title":"Risk-Sensitive Compact Decision Trees for Autonomous Execution in Presence of Simulated Market Response","date":"2019-06-05","arxiv_id":"1906.02312","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-with-low-complexity","slug":"reinforcement-learning-with-low-complexity","title":"Reinforcement Learning with Low-Complexity Liquid State Machines","date":"2019-06-04","arxiv_id":"1906.01695","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wponghiran/lsm-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/190600949","slug":"190600949","title":"Stabilizing Off-Policy Q-Learning via Bootstrapping Error Reduction","date":"2019-06-03","arxiv_id":"1906.00949","n_code_links":3,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":null}},{"paper":null,"slug":"analysis-and-improvement-of-adversarial","title":"Analysis and Improvement of Adversarial Training in DQN Agents With Adversarially-Guided Exploration (AGE)","date":"2019-06-03","arxiv_id":"1906.01119","n_code_links":0,"syntology":null},{"paper":null,"slug":"rl-based-method-for-benchmarking-the","title":"RL-Based Method for Benchmarking the Adversarial Resilience and Robustness of Deep Reinforcement Learning Policies","date":"2019-06-03","arxiv_id":"1906.01110","n_code_links":0,"syntology":null},{"paper":null,"slug":"sequential-triggers-for-watermarking-of-deep","title":"Sequential Triggers for Watermarking of Deep Reinforcement Learning Policies","date":"2019-06-03","arxiv_id":"1906.01126","n_code_links":0,"syntology":null},{"paper":null,"slug":"190600423","title":"Feature-Based Q-Learning for Two-Player Stochastic Games","date":"2019-06-02","arxiv_id":"1906.00423","n_code_links":0,"syntology":null},{"paper":null,"slug":"provably-efficient-q-learning-with-low","title":"Provably Efficient Q-Learning with Low Switching Cost","date":"2019-05-30","arxiv_id":"1905.12849","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-for-slate-based","slug":"reinforcement-learning-for-slate-based","title":"Reinforcement Learning for Slate-based Recommender Systems: A Tractable Decomposition and Practical Methodology","date":"2019-05-29","arxiv_id":"1905.12767","n_code_links":3,"syntology":null},{"paper":null,"slug":"learning-distant-cause-and-effect-using-only","title":"Learning distant cause and effect using only local and immediate credit assignment","date":"2019-05-28","arxiv_id":"1905.11589","n_code_links":0,"syntology":null},{"paper":"/paper/solving-np-hard-problems-on-graphs-by","slug":"solving-np-hard-problems-on-graphs-by","title":"Solving NP-Hard Problems on Graphs with Extended AlphaGo Zero","date":"2019-05-28","arxiv_id":"1905.11623","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/finite-time-analysis-of-q-learning-with","slug":"finite-time-analysis-of-q-learning-with","title":"Finite-Sample Analysis of Nonlinear Stochastic Approximation with Applications in Reinforcement Learning","date":"2019-05-27","arxiv_id":"1905.11425","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/sqil-imitation-learning-via-regularized","slug":"sqil-imitation-learning-via-regularized","title":"SQIL: Imitation Learning via Reinforcement Learning with Sparse Rewards","date":"2019-05-27","arxiv_id":"1905.11108","n_code_links":5,"syntology":null},{"paper":null,"slug":"190512726","title":"Prioritized Sequence Experience Replay","date":"2019-05-25","arxiv_id":"1905.12726","n_code_links":0,"syntology":null},{"paper":"/paper/a-kernel-loss-for-solving-the-bellman","slug":"a-kernel-loss-for-solving-the-bellman","title":"A Kernel Loss for Solving the Bellman Equation","date":"2019-05-25","arxiv_id":"1905.10506","n_code_links":1,"syntology":null},{"paper":null,"slug":"190512567","title":"MQLV: Optimal Policy of Money Management in Retail Banking with Q-Learning","date":"2019-05-24","arxiv_id":"1905.12567","n_code_links":0,"syntology":null},{"paper":"/paper/adaptive-symmetric-reward-noising-for","slug":"adaptive-symmetric-reward-noising-for","title":"Adaptive Symmetric Reward Noising for Reinforcement Learning","date":"2019-05-24","arxiv_id":"1905.10144","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-q-learning-with-q-matrix-transfer","title":"Deep Q-Learning with Q-Matrix Transfer Learning for Novel Fire Evacuation Environment","date":"2019-05-23","arxiv_id":"1905.09673","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-based-parameter","slug":"deep-reinforcement-learning-based-parameter","title":"Deep Reinforcement Learning Based Parameter Control in Differential Evolution","date":"2019-05-20","arxiv_id":"1905.08006","n_code_links":1,"syntology":null},{"paper":null,"slug":"stochastic-variance-reduction-for-deep-q","title":"Stochastic Variance Reduction for Deep Q-learning","date":"2019-05-20","arxiv_id":"1905.08152","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-learning-of","title":"Reinforcement Learning for Learning of Dynamical Systems in Uncertain Environment: a Tutorial","date":"2019-05-19","arxiv_id":"1905.07727","n_code_links":0,"syntology":null},{"paper":"/paper/mastering-the-game-of-sungka-from-random-play","slug":"mastering-the-game-of-sungka-from-random-play","title":"Mastering the Game of Sungka from Random Play","date":"2019-05-17","arxiv_id":"1905.07102","n_code_links":1,"syntology":null},{"paper":"/paper/qbso-fs-a-reinforcement-learning-based-bee","slug":"qbso-fs-a-reinforcement-learning-based-bee","title":"QBSO-FS: A Reinforcement Learning Based Bee Swarm Optimization Metaheuristic for Feature Selection","date":"2019-05-16","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"autonomous-penetration-testing-using","title":"Autonomous Penetration Testing using Reinforcement Learning","date":"2019-05-15","arxiv_id":"1905.05965","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-robotics-and","title":"Reinforcement Learning for Robotics and Control with Active Uncertainty Reduction","date":"2019-05-15","arxiv_id":"1905.06274","n_code_links":0,"syntology":null},{"paper":null,"slug":"design-of-artificial-intelligence-agents-for","title":"Design of Artificial Intelligence Agents for Games using Deep Reinforcement Learning","date":"2019-05-10","arxiv_id":"1905.04127","n_code_links":0,"syntology":null},{"paper":null,"slug":"domain-adversarial-reinforcement-learning-for","title":"Domain Adversarial Reinforcement Learning for Partial Domain Adaptation","date":"2019-05-10","arxiv_id":"1905.04094","n_code_links":0,"syntology":null},{"paper":null,"slug":"190503501","title":"Pretrain Soft Q-Learning with Imperfect Demonstrations","date":"2019-05-09","arxiv_id":"1905.03501","n_code_links":0,"syntology":null},{"paper":null,"slug":"190503726","title":"A Reinforcement Learning Perspective on the Optimal Control of Mutation Probabilities for the (1+1) Evolutionary Algorithm: First Results on the OneMax Problem","date":"2019-05-09","arxiv_id":"1905.03726","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerated-target-updates-for-q-learning","title":"Accelerated Target Updates for Q-learning","date":"2019-05-07","arxiv_id":"1905.02841","n_code_links":0,"syntology":null},{"paper":"/paper/comprehensible-context-driven-text-game","slug":"comprehensible-context-driven-text-game","title":"Comprehensible Context-driven Text Game Playing","date":"2019-05-06","arxiv_id":"1905.02265","n_code_links":2,"syntology":null},{"paper":"/paper/deep-ordinal-reinforcement-learning","slug":"deep-ordinal-reinforcement-learning","title":"Deep Ordinal Reinforcement Learning","date":"2019-05-06","arxiv_id":"1905.02005","n_code_links":1,"syntology":null},{"paper":null,"slug":"beyond-games-bringing-exploration-to-robots","title":"Beyond Games: Bringing Exploration to Robots in Real-world","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/efficient-model-free-reinforcement-learning-1","slug":"efficient-model-free-reinforcement-learning-1","title":"Efficient Model-free Reinforcement Learning in Metric Spaces","date":"2019-05-01","arxiv_id":"1905.00475","n_code_links":1,"syntology":null},{"paper":null,"slug":"inducing-cooperation-via-learning-to-reshape","title":"Inducing Cooperation via Learning to reshape rewards in semi-cooperative multi-agent reinforcement learning","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-agents-with-prioritization-and","title":"Learning agents with prioritization and parameter noise in continuous state and action space","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/recurrent-experience-replay-in-distributed","slug":"recurrent-experience-replay-in-distributed","title":"Recurrent Experience Replay in Distributed Reinforcement Learning","date":"2019-05-01","arxiv_id":null,"n_code_links":3,"syntology":null},{"paper":null,"slug":"a-deep-q-learning-method-for-downlink-power","title":"A Deep Q-Learning Method for Downlink Power Allocation in Multi-Cell Networks","date":"2019-04-30","arxiv_id":"1904.13032","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-adversarial-imagination-for-sample","title":"Generative Adversarial Imagination for Sample Efficient Deep Reinforcement Learning","date":"2019-04-30","arxiv_id":"1904.13255","n_code_links":0,"syntology":null},{"paper":null,"slug":"zapq-learning-for-optimal-stopping-time","title":"Zap Q-Learning for Optimal Stopping Time Problems","date":"2019-04-25","arxiv_id":"1904.11538","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-lipschitz-q-learning","title":"Stochastic Lipschitz Q-Learning","date":"2019-04-24","arxiv_id":"1904.10653","n_code_links":0,"syntology":null},{"paper":null,"slug":"target-based-temporal-difference-learning","title":"Target-Based Temporal Difference Learning","date":"2019-04-24","arxiv_id":"1904.10945","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-q-learning-driven-ct-pancreas","title":"Deep Q Learning Driven CT Pancreas Segmentation with Geometry-Aware U-Net","date":"2019-04-19","arxiv_id":"1904.09120","n_code_links":0,"syntology":null},{"paper":null,"slug":"jam-me-if-you-can-defeating-jammer-with-deep","title":"\"Jam Me If You Can'': Defeating Jammer with Deep Dueling Neural Network Architecture and Ambient Backscattering Augmented Communications","date":"2019-04-08","arxiv_id":"1904.03897","n_code_links":0,"syntology":null},{"paper":null,"slug":"personalized-cancer-chemotherapy-schedule-a","title":"Personalized Cancer Chemotherapy Schedule: a numerical comparison of performance and robustness in model-based and model-free scheduling methodologies","date":"2019-04-02","arxiv_id":"1904.01200","n_code_links":0,"syntology":null},{"paper":null,"slug":"lane-change-decision-making-through-deep","title":"Lane Change Decision-making through Deep Reinforcement Learning with Rule-based Constraints","date":"2019-03-30","arxiv_id":"1904.00231","n_code_links":0,"syntology":null},{"paper":"/paper/improved-robustness-of-reinforcement-learning","slug":"improved-robustness-of-reinforcement-learning","title":"Improved robustness of reinforcement learning policies upon conversion to spiking neuronal network platforms applied to ATARI games","date":"2019-03-26","arxiv_id":"1903.11012","n_code_links":3,"syntology":null},{"paper":null,"slug":"q-learning-for-continuous-actions-with-cross","title":"Q-Learning for Continuous Actions with Cross-Entropy Guided Policies","date":"2019-03-25","arxiv_id":"1903.10605","n_code_links":0,"syntology":null},{"paper":null,"slug":"dqn-with-model-based-exploration-efficient","title":"DQN with model-based exploration: efficient learning on environments with sparse rewards","date":"2019-03-22","arxiv_id":"1903.09295","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-characterizing-divergence-in-deep-q","title":"Towards Characterizing Divergence in Deep Q-Learning","date":"2019-03-21","arxiv_id":"1903.08894","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-with","title":"Deep Reinforcement Learning with Decorrelation","date":"2019-03-18","arxiv_id":"1903.07765","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-with-dynamic-boltzmann","slug":"reinforcement-learning-with-dynamic-boltzmann","title":"Reinforcement Learning with Dynamic Boltzmann Softmax Updates","date":"2019-03-14","arxiv_id":"1903.05926","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-multi-agent-reinforcement-learning-with-1","title":"Deep Multi-Agent Reinforcement Learning with Discrete-Continuous Hybrid Action Spaces","date":"2019-03-12","arxiv_id":"1903.04959","n_code_links":0,"syntology":null},{"paper":"/paper/deep-recurrent-q-learning-vs-deep-q-learning","slug":"deep-recurrent-q-learning-vs-deep-q-learning","title":"Deep Recurrent Q-Learning vs Deep Q-Learning on a simple Partially Observable Markov Decision Process with Minecraft","date":"2019-03-11","arxiv_id":"1903.04311","n_code_links":2,"syntology":null},{"paper":"/paper/multi-agent-deep-reinforcement-learning-for-2","slug":"multi-agent-deep-reinforcement-learning-for-2","title":"Multi-Agent Deep Reinforcement Learning for Large-scale Traffic Signal Control","date":"2019-03-11","arxiv_id":"1903.04527","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":6,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cts198859/deeprl_signal_control"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/sample-efficient-model-free-reinforcement","slug":"sample-efficient-model-free-reinforcement","title":"Sample-Efficient Model-Free Reinforcement Learning with Off-Policy Critics","date":"2019-03-11","arxiv_id":"1903.04193","n_code_links":1,"syntology":null},{"paper":null,"slug":"deeppool-distributed-model-free-algorithm-for","title":"DeepPool: Distributed Model-free Algorithm for Ride-sharing using Deep Reinforcement Learning","date":"2019-03-09","arxiv_id":"1903.03882","n_code_links":0,"syntology":null},{"paper":null,"slug":"successive-over-relaxation-q-learning","title":"Successive Over Relaxation Q-Learning","date":"2019-03-09","arxiv_id":"1903.03812","n_code_links":0,"syntology":null},{"paper":"/paper/learning-heuristics-over-large-graphs-via","slug":"learning-heuristics-over-large-graphs-via","title":"Learning Heuristics over Large Graphs via Deep Reinforcement Learning","date":"2019-03-08","arxiv_id":"1903.03332","n_code_links":2,"syntology":null},{"paper":"/paper/minatar-an-atari-inspired-testbed-for-more","slug":"minatar-an-atari-inspired-testbed-for-more","title":"MinAtar: An Atari-Inspired Testbed for Thorough and Reproducible Reinforcement Learning Experiments","date":"2019-03-07","arxiv_id":"1903.03176","n_code_links":3,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["kenjyoung/MinAtar"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"distributed-edge-caching-via-reinforcement","title":"Distributed Edge Caching via Reinforcement Learning in Fog Radio Access Networks","date":"2019-02-27","arxiv_id":"1902.10574","n_code_links":0,"syntology":null},{"paper":null,"slug":"unifying-ensemble-methods-for-q-learning-via","title":"Unifying Ensemble Methods for Q-learning via Social Choice Theory","date":"2019-02-27","arxiv_id":"1902.10646","n_code_links":0,"syntology":null},{"paper":"/paper/diagnosing-bottlenecks-in-deep-q-learning","slug":"diagnosing-bottlenecks-in-deep-q-learning","title":"Diagnosing Bottlenecks in Deep Q-learning Algorithms","date":"2019-02-26","arxiv_id":"1902.10250","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"optimal-and-fast-real-time-resources-slicing","title":"Optimal and Fast Real-time Resources Slicing with Deep Dueling Neural Networks","date":"2019-02-26","arxiv_id":"1902.09696","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-the-next-generation-airline-revenue","title":"Autonomous Airline Revenue Management: A Deep Reinforcement Learning Approach to Seat Inventory Control and Overbooking","date":"2019-02-18","arxiv_id":"1902.06824","n_code_links":0,"syntology":null},{"paper":"/paper/heuristics-answer-set-programming-and-markov","slug":"heuristics-answer-set-programming-and-markov","title":"Heuristics, Answer Set Programming and Markov Decision Process for Solving a Set of Spatial Puzzles","date":"2019-02-16","arxiv_id":"1903.03411","n_code_links":1,"syntology":null},{"paper":null,"slug":"sample-optimal-parametric-q-learning-with","title":"Sample-Optimal Parametric Q-Learning Using Linearly Additive Features","date":"2019-02-13","arxiv_id":"1902.04779","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-best-response-strategies-for-agents","title":"Learning Best Response Strategies for Agents in Ad Exchanges","date":"2019-02-10","arxiv_id":"1902.03588","n_code_links":0,"syntology":null},{"paper":"/paper/making-deep-q-learning-methods-robust-to-time","slug":"making-deep-q-learning-methods-robust-to-time","title":"Making Deep Q-learning methods robust to time discretization","date":"2019-01-28","arxiv_id":"1901.09732","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"q-learning-with-ucb-exploration-is-sample","title":"Q-learning with UCB Exploration is Sample Efficient for Infinite-Horizon MDP","date":"2019-01-27","arxiv_id":"1901.09311","n_code_links":0,"syntology":null},{"paper":null,"slug":"reward-shaping-via-meta-learning","title":"Reward Shaping via Meta-Learning","date":"2019-01-27","arxiv_id":"1901.09330","n_code_links":0,"syntology":null},{"paper":"/paper/combinational-q-learning-for-dou-di-zhu","slug":"combinational-q-learning-for-dou-di-zhu","title":"Combinational Q-Learning for Dou Di Zhu","date":"2019-01-24","arxiv_id":"1901.08925","n_code_links":1,"syntology":null},{"paper":null,"slug":"distillation-strategies-for-proximal-policy","title":"Distillation Strategies for Proximal Policy Optimization","date":"2019-01-23","arxiv_id":"1901.08128","n_code_links":0,"syntology":null}],"record_sha256":"93f379a0e0262eb8065dc2cf73c498d9e22389d4fbf71839afabf3b6a517fd86","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}