{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/17","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":17,"pages_in_order":18,"rows_per_page":100,"rows":[1601,1700],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning/papers/16","next":"/method/q-learning/papers/18","papers":[{"paper":"/paper/monte-carlo-q-learning-for-general-game","slug":"monte-carlo-q-learning-for-general-game","title":"Monte Carlo Q-learning for General Game Playing","date":"2018-02-16","arxiv_id":"1802.05944","n_code_links":2,"syntology":null},{"paper":"/paper/mean-field-multi-agent-reinforcement-learning","slug":"mean-field-multi-agent-reinforcement-learning","title":"Mean Field Multi-Agent Reinforcement Learning","date":"2018-02-15","arxiv_id":"1802.05438","n_code_links":3,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mlii/mfrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/efficient-exploration-through-bayesian-deep-q","slug":"efficient-exploration-through-bayesian-deep-q","title":"Efficient Exploration through Bayesian Deep Q-Networks","date":"2018-02-13","arxiv_id":"1802.04412","n_code_links":1,"syntology":null},{"paper":null,"slug":"q-learning-with-nearest-neighbors","title":"Q-learning with Nearest Neighbors","date":"2018-02-12","arxiv_id":"1802.03900","n_code_links":0,"syntology":null},{"paper":null,"slug":"balancing-two-player-stochastic-games-with","title":"Balancing Two-Player Stochastic Games with Soft Q-Learning","date":"2018-02-09","arxiv_id":"1802.03216","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-using-capsules-in","title":"Deep Reinforcement Learning using Capsules in Advanced Game Environments","date":"2018-01-29","arxiv_id":"1801.09597","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-qlbs-q-learner-goes-nuqlear-fitted-q","title":"The QLBS Q-Learner Goes NuQLear: Fitted Q Iteration, Inverse RL, and Option Portfolios","date":"2018-01-17","arxiv_id":"1801.06077","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-fuzzing","title":"Deep Reinforcement Fuzzing","date":"2018-01-14","arxiv_id":"1801.04589","n_code_links":0,"syntology":null},{"paper":"/paper/deeptraffic-crowdsourced-hyperparameter","slug":"deeptraffic-crowdsourced-hyperparameter","title":"DeepTraffic: Crowdsourced Hyperparameter Tuning of Deep Reinforcement Learning Systems for Multi-Agent Dense Traffic Navigation","date":"2018-01-09","arxiv_id":"1801.02805","n_code_links":6,"syntology":null},{"paper":null,"slug":"faster-deep-q-learning-using-neural-episodic","title":"Faster Deep Q-learning using Neural Episodic Control","date":"2018-01-06","arxiv_id":"1801.01968","n_code_links":0,"syntology":null},{"paper":null,"slug":"autonomous-vehicle-fleet-coordination-with","title":"Autonomous Vehicle Fleet Coordination With Deep Reinforcement Learning","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"avoiding-catastrophic-states-with-intrinsic","title":"Avoiding Catastrophic States with Intrinsic Fear","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"faster-reinforcement-learning-with-expert","title":"Faster Reinforcement Learning with Expert State Sequences","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/parametrized-deep-q-networks-learning-playing","slug":"parametrized-deep-q-networks-learning-playing","title":"PARAMETRIZED DEEP Q-NETWORKS LEARNING: PLAYING ONLINE BATTLE ARENA WITH DISCRETE-CONTINUOUS HYBRID ACTION SPACE","date":"2018-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"representing-entropy-a-short-proof-of-the","title":"Representing Entropy : A short proof of the equivalence between soft Q-learning and policy gradients","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"td-learning-with-constrained-gradients","title":"TD Learning with Constrained Gradients","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-short-variational-proof-of-equivalence","title":"A short variational proof of equivalence between policy gradients and soft Q learning","date":"2017-12-22","arxiv_id":"1712.08650","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deep-policy-inference-q-network-for-multi","title":"A Deep Policy Inference Q-Network for Multi-Agent Systems","date":"2017-12-21","arxiv_id":"1712.07893","n_code_links":0,"syntology":null},{"paper":"/paper/deep-neuroevolution-genetic-algorithms-are-a","slug":"deep-neuroevolution-genetic-algorithms-are-a","title":"Deep Neuroevolution: Genetic Algorithms Are a Competitive Alternative for Training Deep Neural Networks for Reinforcement Learning","date":"2017-12-18","arxiv_id":"1712.06567","n_code_links":12,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/improving-exploration-in-evolution-strategies","slug":"improving-exploration-in-evolution-strategies","title":"Improving Exploration in Evolution Strategies for Deep Reinforcement Learning via a Population of Novelty-Seeking Agents","date":"2017-12-18","arxiv_id":"1712.06560","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["uber-research/deep-neuroevolution"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"towards-a-deep-reinforcement-learning","title":"Towards a Deep Reinforcement Learning Approach for Tower Line Wars","date":"2017-12-17","arxiv_id":"1712.06180","n_code_links":0,"syntology":null},{"paper":"/paper/qlbs-q-learner-in-the-black-scholes-merton","slug":"qlbs-q-learner-in-the-black-scholes-merton","title":"QLBS: Q-Learner in the Black-Scholes(-Merton) Worlds","date":"2017-12-13","arxiv_id":"1712.04609","n_code_links":1,"syntology":null},{"paper":"/paper/assumed-density-filtering-q-learning","slug":"assumed-density-filtering-q-learning","title":"Assumed Density Filtering Q-learning","date":"2017-12-09","arxiv_id":"1712.03333","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-primal-dual-reinforcement-learning","title":"Deep Primal-Dual Reinforcement Learning: Accelerating Actor-Critic using Bellman Duality","date":"2017-12-07","arxiv_id":"1712.02467","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-lda-uncovering-latent-patterns-in-text","title":"Q-LDA: Uncovering Latent Patterns in Text-based Sequential Decision Processes","date":"2017-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"zap-q-learning","title":"Zap Q-Learning","date":"2017-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"uncertainty-estimates-for-efficient-neural","title":"Uncertainty Estimates for Efficient Neural Network-based Dialogue Policy Optimisation","date":"2017-11-30","arxiv_id":"1711.11486","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-benchmarking-environment-for-reinforcement","title":"A Benchmarking Environment for Reinforcement Learning Based Task Oriented Dialogue Management","date":"2017-11-29","arxiv_id":"1711.11023","n_code_links":0,"syntology":null},{"paper":"/paper/implementing-the-deep-q-network","slug":"implementing-the-deep-q-network","title":"Implementing the Deep Q-Network","date":"2017-11-20","arxiv_id":"1711.07478","n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-network-based-reinforcement-learning","title":"Neural Network Based Reinforcement Learning for Audio-Visual Gaze Control in Human-Robot Interaction","date":"2017-11-18","arxiv_id":"1711.06834","n_code_links":0,"syntology":null},{"paper":null,"slug":"bbq-networks-efficient-exploration-in-deep","title":"BBQ-Networks: Efficient Exploration in Deep Reinforcement Learning for Task-Oriented Dialogue Systems","date":"2017-11-15","arxiv_id":"1711.05715","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-unified-decision-making-framework-for","title":"A unified decision making framework for supply and demand management in microgrid networks","date":"2017-11-14","arxiv_id":"1711.05078","n_code_links":0,"syntology":null},{"paper":null,"slug":"double-q-and-q-unifying-reinforcement","title":"Double Q($σ$) and Q($σ, λ$): Unifying Reinforcement Learning Control Algorithms","date":"2017-11-05","arxiv_id":"1711.01569","n_code_links":0,"syntology":null},{"paper":"/paper/adaptive-coordination-of-working-memory-and","slug":"adaptive-coordination-of-working-memory-and","title":"Adaptive coordination of working-memory and reinforcement learning in non-human primates performing a trial-and-error problem solving task","date":"2017-11-02","arxiv_id":"1711.00698","n_code_links":1,"syntology":null},{"paper":"/paper/treeqn-and-atreec-differentiable-tree","slug":"treeqn-and-atreec-differentiable-tree","title":"TreeQN and ATreeC: Differentiable Tree-Structured Models for Deep Reinforcement Learning","date":"2017-10-31","arxiv_id":"1710.11417","n_code_links":1,"syntology":{"ran":7,"of":12,"n_ran_checked":7,"n_instrument":0,"unverified":5,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["oxwhirl/treeqn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/distributional-reinforcement-learning-with-1","slug":"distributional-reinforcement-learning-with-1","title":"Distributional Reinforcement Learning with Quantile Regression","date":"2017-10-27","arxiv_id":"1710.10044","n_code_links":17,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/the-effects-of-memory-replay-in-reinforcement","slug":"the-effects-of-memory-replay-in-reinforcement","title":"The Effects of Memory Replay in Reinforcement Learning","date":"2017-10-18","arxiv_id":"1710.06574","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-framework","title":"Deep Reinforcement Learning: Framework, Applications, and Embedded Implementations","date":"2017-10-10","arxiv_id":"1710.03792","n_code_links":0,"syntology":null},{"paper":"/paper/rainbow-combining-improvements-in-deep","slug":"rainbow-combining-improvements-in-deep","title":"Rainbow: Combining Improvements in Deep Reinforcement Learning","date":"2017-10-06","arxiv_id":"1710.02298","n_code_links":34,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"supervised-q-walk-for-learning-vector","title":"Supervised Q-walk for Learning Vector Representation of Nodes in Networks","date":"2017-10-03","arxiv_id":"1710.00978","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-simple-reinforcement-learning-mechanism-for","title":"A Simple Reinforcement Learning Mechanism for Resource Allocation in LTE-A Networks with Markov Decision Process and Q-Learning","date":"2017-09-27","arxiv_id":"1709.09312","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-optimal-online-method-of-selecting-source","title":"An Optimal Online Method of Selecting Source Policies for Reinforcement Learning","date":"2017-09-24","arxiv_id":"1709.08201","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-search-through-a3c-reinforcement","title":"Improving Search through A3C Reinforcement Learning based Conversational Agent","date":"2017-09-17","arxiv_id":"1709.05638","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-with-surrogate","title":"Deep Reinforcement Learning with Surrogate Agent-Environment Interface","date":"2017-09-12","arxiv_id":"1709.03942","n_code_links":0,"syntology":null},{"paper":null,"slug":"pre-training-neural-networks-with-human","title":"Pre-training Neural Networks with Human Demonstrations for Deep Reinforcement Learning","date":"2017-09-12","arxiv_id":"1709.04083","n_code_links":0,"syntology":null},{"paper":null,"slug":"formulation-of-deep-reinforcement-learning","title":"Formulation of Deep Reinforcement Learning Architecture Toward Autonomous Driving for On-Ramp Merge","date":"2017-09-07","arxiv_id":"1709.02066","n_code_links":0,"syntology":null},{"paper":null,"slug":"bibi-system-description-building-with-cnns","title":"BIBI System Description: Building with CNNs and Breaking with Deep Reinforcement Learning","date":"2017-09-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-agent-q-learning-for-minimizing-demand","title":"Multi-Agent Q-Learning for Minimizing Demand-Supply Power Deficit in Microgrids","date":"2017-08-25","arxiv_id":"1708.07732","n_code_links":0,"syntology":null},{"paper":null,"slug":"ladder-a-human-level-bidding-agent-for-large","title":"LADDER: A Human-Level Bidding Agent for Large-Scale Real-Time Online Auctions","date":"2017-08-18","arxiv_id":"1708.05565","n_code_links":0,"syntology":null},{"paper":"/paper/practical-block-wise-neural-network","slug":"practical-block-wise-neural-network","title":"Practical Block-wise Neural Network Architecture Generation","date":"2017-08-18","arxiv_id":"1708.05552","n_code_links":1,"syntology":null},{"paper":null,"slug":"investigating-reinforcement-learning-agents","title":"Investigating Reinforcement Learning Agents for Continuous State Space Environments","date":"2017-08-08","arxiv_id":"1708.02378","n_code_links":0,"syntology":null},{"paper":null,"slug":"3dcnn-dqn-rnn-a-deep-reinforcement-learning","title":"3DCNN-DQN-RNN: A Deep Reinforcement Learning Framework for Semantic Parsing of Large-scale 3D Point Clouds","date":"2017-07-21","arxiv_id":"1707.06783","n_code_links":0,"syntology":null},{"paper":null,"slug":"empirical-evaluation-of-a-q-learning","title":"Empirical evaluation of a Q-Learning Algorithm for Model-free Autonomous Soaring","date":"2017-07-18","arxiv_id":"1707.05668","n_code_links":0,"syntology":null},{"paper":null,"slug":"fastest-convergence-for-q-learning","title":"Fastest Convergence for Q-learning","date":"2017-07-12","arxiv_id":"1707.03770","n_code_links":0,"syntology":null},{"paper":"/paper/noisy-networks-for-exploration","slug":"noisy-networks-for-exploration","title":"Noisy Networks for Exploration","date":"2017-06-30","arxiv_id":"1706.10295","n_code_links":15,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/a-self-adaptive-proposal-model-for-temporal","slug":"a-self-adaptive-proposal-model-for-temporal","title":"A Self-Adaptive Proposal Model for Temporal Action Detection based on Reinforcement Learning","date":"2017-06-22","arxiv_id":"1706.07251","n_code_links":1,"syntology":null},{"paper":"/paper/generalized-value-iteration-networks-life","slug":"generalized-value-iteration-networks-life","title":"Generalized Value Iteration Networks: Life Beyond Lattices","date":"2017-06-08","arxiv_id":"1706.02416","n_code_links":1,"syntology":null},{"paper":"/paper/multi-agent-actor-critic-for-mixed","slug":"multi-agent-actor-critic-for-mixed","title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","date":"2017-06-07","arxiv_id":"1706.02275","n_code_links":86,"syntology":{"ran":75,"of":143,"n_ran_checked":68,"n_instrument":7,"unverified":68,"pointer_only":99,"phrase":"75 ran (of which 54 constructed an object rather than computing a result; 68 with no instrument failure: 2 honoured, 0 violated, 66 with no contract checked; 7 where Syntology's instrument failed) · 68 unverified","official":{"repos":["openai/multiagent-particle-envs"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/parameter-space-noise-for-exploration","slug":"parameter-space-noise-for-exploration","title":"Parameter Space Noise for Exploration","date":"2017-06-06","arxiv_id":"1706.01905","n_code_links":10,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"explaining-transition-systems-through-program","title":"Explaining Transition Systems through Program Induction","date":"2017-05-23","arxiv_id":"1705.08320","n_code_links":0,"syntology":null},{"paper":null,"slug":"shallow-updates-for-deep-reinforcement","title":"Shallow Updates for Deep Reinforcement Learning","date":"2017-05-21","arxiv_id":"1705.07461","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparison-of-reinforcement-learning","title":"A Comparison of Reinforcement Learning Techniques for Fuzzy Cloud Auto-Scaling","date":"2017-05-19","arxiv_id":"1705.07114","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-represent-haptic-feedback-for","title":"Learning to Represent Haptic Feedback for Partially-Observable Tasks","date":"2017-05-17","arxiv_id":"1705.06243","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-hard-alignments-with-variational","title":"Learning Hard Alignments with Variational Inference","date":"2017-05-16","arxiv_id":"1705.05524","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-episodic-value-iteration-for-model-based","title":"Deep Episodic Value Iteration for Model-based Meta-Reinforcement Learning","date":"2017-05-09","arxiv_id":"1705.03562","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-with-external","title":"Reinforcement Learning with External Knowledge and Two-Stage Q-functions for Predicting Popular Reddit Threads","date":"2017-04-20","arxiv_id":"1704.06217","n_code_links":0,"syntology":null},{"paper":"/paper/the-reactor-a-fast-and-sample-efficient-actor","slug":"the-reactor-a-fast-and-sample-efficient-actor","title":"The Reactor: A fast and sample-efficient Actor-Critic agent for Reinforcement Learning","date":"2017-04-15","arxiv_id":"1704.04651","n_code_links":0,"syntology":null},{"paper":"/paper/deep-q-learning-from-demonstrations","slug":"deep-q-learning-from-demonstrations","title":"Deep Q-learning from Demonstrations","date":"2017-04-12","arxiv_id":"1704.03732","n_code_links":6,"syntology":null},{"paper":null,"slug":"data-efficient-deep-reinforcement-learning","title":"Data-efficient Deep Reinforcement Learning for Dexterous Manipulation","date":"2017-04-10","arxiv_id":"1704.03073","n_code_links":0,"syntology":null},{"paper":null,"slug":"pseudorehearsal-in-value-function","title":"Pseudorehearsal in value function approximation","date":"2017-03-21","arxiv_id":"1703.07075","n_code_links":0,"syntology":null},{"paper":"/paper/evolution-strategies-as-a-scalable","slug":"evolution-strategies-as-a-scalable","title":"Evolution Strategies as a Scalable Alternative to Reinforcement Learning","date":"2017-03-10","arxiv_id":"1703.03864","n_code_links":23,"syntology":{"ran":16,"of":29,"n_ran_checked":11,"n_instrument":5,"unverified":13,"pointer_only":2,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 5 where Syntology's instrument failed) · 13 unverified","official":{"repos":["openai/evolution-strategies-starter"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":null,"slug":"tactics-of-adversarial-attack-on-deep","title":"Tactics of Adversarial Attack on Deep Reinforcement Learning Agents","date":"2017-03-08","arxiv_id":"1703.06748","n_code_links":0,"syntology":null},{"paper":"/paper/count-based-exploration-with-neural-density","slug":"count-based-exploration-with-neural-density","title":"Count-Based Exploration with Neural Density Models","date":"2017-03-03","arxiv_id":"1703.01310","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/bridging-the-gap-between-value-and-policy","slug":"bridging-the-gap-between-value-and-policy","title":"Bridging the Gap Between Value and Policy Based Reinforcement Learning","date":"2017-02-28","arxiv_id":"1702.08892","n_code_links":1,"syntology":null},{"paper":"/paper/stabilising-experience-replay-for-deep-multi","slug":"stabilising-experience-replay-for-deep-multi","title":"Stabilising Experience Replay for Deep Multi-Agent Reinforcement Learning","date":"2017-02-28","arxiv_id":"1702.08887","n_code_links":5,"syntology":null},{"paper":null,"slug":"learning-control-for-air-hockey-striking","title":"Learning Control for Air Hockey Striking using Deep Reinforcement Learning","date":"2017-02-26","arxiv_id":"1702.08074","n_code_links":0,"syntology":null},{"paper":"/paper/sigmoid-weighted-linear-units-for-neural","slug":"sigmoid-weighted-linear-units-for-neural","title":"Sigmoid-Weighted Linear Units for Neural Network Function Approximation in Reinforcement Learning","date":"2017-02-10","arxiv_id":"1702.03118","n_code_links":0,"syntology":null},{"paper":"/paper/autonomous-braking-system-via-deep","slug":"autonomous-braking-system-via-deep","title":"Autonomous Braking System via Deep Reinforcement Learning","date":"2017-02-08","arxiv_id":"1702.02302","n_code_links":2,"syntology":null},{"paper":null,"slug":"fpga-architecture-for-deep-learning-and-its","title":"FPGA Architecture for Deep Learning and its application to Planetary Robotics","date":"2017-01-26","arxiv_id":"1701.07543","n_code_links":0,"syntology":null},{"paper":"/paper/vulnerability-of-deep-reinforcement-learning","slug":"vulnerability-of-deep-reinforcement-learning","title":"Vulnerability of Deep Reinforcement Learning to Policy Induction Attacks","date":"2017-01-16","arxiv_id":"1701.04143","n_code_links":1,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-for-multi-domain","slug":"deep-reinforcement-learning-for-multi-domain","title":"Deep Reinforcement Learning for Multi-Domain Dialogue Systems","date":"2016-11-26","arxiv_id":"1611.08675","n_code_links":1,"syntology":null},{"paper":null,"slug":"memory-lens-how-much-memory-does-an-agent-use","title":"Memory Lens: How Much Memory Does an Agent Use?","date":"2016-11-21","arxiv_id":"1611.06928","n_code_links":0,"syntology":null},{"paper":null,"slug":"averaged-dqn-variance-reduction-and","title":"Averaged-DQN: Variance Reduction and Stabilization for Deep Reinforcement Learning","date":"2016-11-07","arxiv_id":"1611.01929","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-play-in-a-day-faster-deep","slug":"learning-to-play-in-a-day-faster-deep","title":"Learning to Play in a Day: Faster Deep Reinforcement Learning by Optimality Tightening","date":"2016-11-05","arxiv_id":"1611.01606","n_code_links":1,"syntology":null},{"paper":null,"slug":"combating-reinforcement-learnings-sisyphean","title":"Combating Reinforcement Learning's Sisyphean Curse with Intrinsic Fear","date":"2016-11-03","arxiv_id":"1611.01211","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-a-deep-reinforcement-learning-agent-for","title":"Using a Deep Reinforcement Learning Agent for Traffic Signal Control","date":"2016-11-03","arxiv_id":"1611.01142","n_code_links":0,"syntology":null},{"paper":null,"slug":"internet-of-things-applications-animal","title":"Internet of Things Applications: Animal Monitoring with Unmanned Aerial Vehicle","date":"2016-10-17","arxiv_id":"1610.05287","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-from-raw-pixels","title":"Deep Reinforcement Learning From Raw Pixels in Doom","date":"2016-10-07","arxiv_id":"1610.02164","n_code_links":0,"syntology":null},{"paper":"/paper/active-exploration-in-parameterized","slug":"active-exploration-in-parameterized","title":"Active exploration in parameterized reinforcement learning","date":"2016-10-06","arxiv_id":"1610.01986","n_code_links":1,"syntology":null},{"paper":"/paper/opponent-modeling-in-deep-reinforcement","slug":"opponent-modeling-in-deep-reinforcement","title":"Opponent Modeling in Deep Reinforcement Learning","date":"2016-09-18","arxiv_id":"1609.05559","n_code_links":1,"syntology":null},{"paper":null,"slug":"episodic-exploration-for-deep-deterministic","title":"Episodic Exploration for Deep Deterministic Policies: An Application to StarCraft Micromanagement Tasks","date":"2016-09-10","arxiv_id":"1609.02993","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-exit-configuration-of-mesoscopic","title":"Multi Exit Configuration of Mesoscopic Pedestrian Simulation","date":"2016-09-06","arxiv_id":"1609.01475","n_code_links":0,"syntology":null},{"paper":null,"slug":"bbq-networks-efficient-exploration-in-deep-1","title":"BBQ-Networks: Efficient Exploration in Deep Reinforcement Learning for Task-Oriented Dialogue Systems","date":"2016-08-17","arxiv_id":"1608.05081","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-discovers","title":"Deep Reinforcement Learning Discovers Internal Models","date":"2016-06-16","arxiv_id":"1606.05174","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-with-macro","title":"Deep Reinforcement Learning With Macro-Actions","date":"2016-06-15","arxiv_id":"1606.04615","n_code_links":0,"syntology":null},{"paper":"/paper/vizdoom-a-doom-based-ai-research-platform-for","slug":"vizdoom-a-doom-based-ai-research-platform-for","title":"ViZDoom: A Doom-based AI Research Platform for Visual Reinforcement Learning","date":"2016-05-06","arxiv_id":"1605.02097","n_code_links":10,"syntology":{"ran":2,"of":5,"n_ran_checked":1,"n_instrument":1,"unverified":3,"pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["mwydmuch/ViZDoom"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"classifying-options-for-deep-reinforcement","title":"Classifying Options for Deep Reinforcement Learning","date":"2016-04-27","arxiv_id":"1604.08153","n_code_links":0,"syntology":null},{"paper":null,"slug":"neurohex-a-deep-q-learning-hex-agent","title":"Neurohex: A Deep Q-learning Hex Agent","date":"2016-04-24","arxiv_id":"1604.07097","n_code_links":0,"syntology":null},{"paper":"/paper/continuous-deep-q-learning-with-model-based","slug":"continuous-deep-q-learning-with-model-based","title":"Continuous Deep Q-Learning with Model-based Acceleration","date":"2016-03-02","arxiv_id":"1603.00748","n_code_links":8,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"reinforcement-learning-approach-for-real-time","title":"Reinforcement Learning approach for Real Time Strategy Games Battle city and S3","date":"2016-02-16","arxiv_id":"1602.04936","n_code_links":0,"syntology":null}],"record_sha256":"d3cd08f06107949a4b0fce7f07044bb2b975233e32e604c02c1a75078ebaa988","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}