{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/dqn/papers/5","list_of":"/method/dqn","method":"DQN","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":5,"pages_in_order":6,"rows_per_page":100,"rows":[401,500],"of":519,"counts":{"archive_papers_tagged":519,"with_a_code_link":173,"where_syntology_ran_a_sample":47,"not_listed_spam_title":0,"listed":519,"listed_where_code_ran":47,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":36,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":36,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/dqn","prev":"/method/dqn/papers/4","next":"/method/dqn/papers/6","papers":[{"paper":"/paper/benchmarking-batch-deep-reinforcement","slug":"benchmarking-batch-deep-reinforcement","title":"Benchmarking Batch Deep Reinforcement Learning Algorithms","date":"2019-10-03","arxiv_id":"1910.01708","n_code_links":5,"syntology":null},{"paper":null,"slug":"ai-assisted-annotator-using-reinforcement","title":"AI Assisted Annotator using Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.02052","n_code_links":0,"syntology":null},{"paper":"/paper/190909902","slug":"190909902","title":"Deep Reinforcement Learning with Modulated Hebbian plus Q Network Architecture","date":"2019-09-21","arxiv_id":"1909.09902","n_code_links":1,"syntology":null},{"paper":null,"slug":"split-deep-q-learning-for-robust-object","title":"Split Deep Q-Learning for Robust Object Singulation","date":"2019-09-17","arxiv_id":"1909.08105","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-reinforcement-learning-based-approach-for","title":"Reinforcement Learning for Joint Optimization of Multiple Rewards","date":"2019-09-06","arxiv_id":"1909.02940","n_code_links":0,"syntology":null},{"paper":"/paper/striving-for-simplicity-in-off-policy-deep","slug":"striving-for-simplicity-in-off-policy-deep","title":"An Optimistic Perspective on Offline Reinforcement Learning","date":"2019-07-10","arxiv_id":"1907.04543","n_code_links":1,"syntology":null},{"paper":"/paper/towards-empathic-deep-q-learning","slug":"towards-empathic-deep-q-learning","title":"Towards Empathic Deep Q-Learning","date":"2019-06-26","arxiv_id":"1906.10918","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bartbussmann/EmpathicDQN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-causal-state-representations-of","title":"Learning Causal State Representations of Partially Observable Environments","date":"2019-06-25","arxiv_id":"1906.10437","n_code_links":0,"syntology":null},{"paper":null,"slug":"analysis-and-improvement-of-adversarial","title":"Analysis and Improvement of Adversarial Training in DQN Agents With Adversarially-Guided Exploration (AGE)","date":"2019-06-03","arxiv_id":"1906.01119","n_code_links":0,"syntology":null},{"paper":null,"slug":"rl-based-method-for-benchmarking-the","title":"RL-Based Method for Benchmarking the Adversarial Resilience and Robustness of Deep Reinforcement Learning Policies","date":"2019-06-03","arxiv_id":"1906.01110","n_code_links":0,"syntology":null},{"paper":null,"slug":"sequential-triggers-for-watermarking-of-deep","title":"Sequential Triggers for Watermarking of Deep Reinforcement Learning Policies","date":"2019-06-03","arxiv_id":"1906.01126","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-distant-cause-and-effect-using-only","title":"Learning distant cause and effect using only local and immediate credit assignment","date":"2019-05-28","arxiv_id":"1905.11589","n_code_links":0,"syntology":null},{"paper":null,"slug":"190512726","title":"Prioritized Sequence Experience Replay","date":"2019-05-25","arxiv_id":"1905.12726","n_code_links":0,"syntology":null},{"paper":"/paper/adaptive-symmetric-reward-noising-for","slug":"adaptive-symmetric-reward-noising-for","title":"Adaptive Symmetric Reward Noising for Reinforcement Learning","date":"2019-05-24","arxiv_id":"1905.10144","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-q-learning-with-q-matrix-transfer","title":"Deep Q-Learning with Q-Matrix Transfer Learning for Novel Fire Evacuation Environment","date":"2019-05-23","arxiv_id":"1905.09673","n_code_links":0,"syntology":null},{"paper":"/paper/mastering-the-game-of-sungka-from-random-play","slug":"mastering-the-game-of-sungka-from-random-play","title":"Mastering the Game of Sungka from Random Play","date":"2019-05-17","arxiv_id":"1905.07102","n_code_links":1,"syntology":null},{"paper":"/paper/comprehensible-context-driven-text-game","slug":"comprehensible-context-driven-text-game","title":"Comprehensible Context-driven Text Game Playing","date":"2019-05-06","arxiv_id":"1905.02265","n_code_links":2,"syntology":null},{"paper":null,"slug":"beyond-games-bringing-exploration-to-robots","title":"Beyond Games: Bringing Exploration to Robots in Real-world","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"inducing-cooperation-via-learning-to-reshape","title":"Inducing Cooperation via Learning to reshape rewards in semi-cooperative multi-agent reinforcement learning","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-agents-with-prioritization-and","title":"Learning agents with prioritization and parameter noise in continuous state and action space","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/recurrent-experience-replay-in-distributed","slug":"recurrent-experience-replay-in-distributed","title":"Recurrent Experience Replay in Distributed Reinforcement Learning","date":"2019-05-01","arxiv_id":null,"n_code_links":3,"syntology":null},{"paper":null,"slug":"generative-adversarial-imagination-for-sample","title":"Generative Adversarial Imagination for Sample Efficient Deep Reinforcement Learning","date":"2019-04-30","arxiv_id":"1904.13255","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-q-learning-driven-ct-pancreas","title":"Deep Q Learning Driven CT Pancreas Segmentation with Geometry-Aware U-Net","date":"2019-04-19","arxiv_id":"1904.09120","n_code_links":0,"syntology":null},{"paper":null,"slug":"personalized-cancer-chemotherapy-schedule-a","title":"Personalized Cancer Chemotherapy Schedule: a numerical comparison of performance and robustness in model-based and model-free scheduling methodologies","date":"2019-04-02","arxiv_id":"1904.01200","n_code_links":0,"syntology":null},{"paper":null,"slug":"lane-change-decision-making-through-deep","title":"Lane Change Decision-making through Deep Reinforcement Learning with Rule-based Constraints","date":"2019-03-30","arxiv_id":"1904.00231","n_code_links":0,"syntology":null},{"paper":null,"slug":"dqn-with-model-based-exploration-efficient","title":"DQN with model-based exploration: efficient learning on environments with sparse rewards","date":"2019-03-22","arxiv_id":"1903.09295","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-with","title":"Deep Reinforcement Learning with Decorrelation","date":"2019-03-18","arxiv_id":"1903.07765","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-with-dynamic-boltzmann","slug":"reinforcement-learning-with-dynamic-boltzmann","title":"Reinforcement Learning with Dynamic Boltzmann Softmax Updates","date":"2019-03-14","arxiv_id":"1903.05926","n_code_links":1,"syntology":null},{"paper":"/paper/sample-efficient-model-free-reinforcement","slug":"sample-efficient-model-free-reinforcement","title":"Sample-Efficient Model-Free Reinforcement Learning with Off-Policy Critics","date":"2019-03-11","arxiv_id":"1903.04193","n_code_links":1,"syntology":null},{"paper":null,"slug":"deeppool-distributed-model-free-algorithm-for","title":"DeepPool: Distributed Model-free Algorithm for Ride-sharing using Deep Reinforcement Learning","date":"2019-03-09","arxiv_id":"1903.03882","n_code_links":0,"syntology":null},{"paper":"/paper/minatar-an-atari-inspired-testbed-for-more","slug":"minatar-an-atari-inspired-testbed-for-more","title":"MinAtar: An Atari-Inspired Testbed for Thorough and Reproducible Reinforcement Learning Experiments","date":"2019-03-07","arxiv_id":"1903.03176","n_code_links":3,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["kenjyoung/MinAtar"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"reward-shaping-via-meta-learning","title":"Reward Shaping via Meta-Learning","date":"2019-01-27","arxiv_id":"1901.09330","n_code_links":0,"syntology":null},{"paper":null,"slug":"distillation-strategies-for-proximal-policy","title":"Distillation Strategies for Proximal Policy Optimization","date":"2019-01-23","arxiv_id":"1901.08128","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-multi-step-deep-reinforcement","slug":"understanding-multi-step-deep-reinforcement","title":"Understanding Multi-Step Deep Reinforcement Learning: A Systematic Study of the DQN Target","date":"2019-01-22","arxiv_id":"1901.07510","n_code_links":1,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-for-imbalanced","slug":"deep-reinforcement-learning-for-imbalanced","title":"Deep Reinforcement Learning for Imbalanced Classification","date":"2019-01-05","arxiv_id":"1901.01379","n_code_links":3,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["linenus/DRL-For-imbalanced-Classification"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"a-theoretical-analysis-of-deep-q-learning","title":"A Theoretical Analysis of Deep Q-Learning","date":"2019-01-01","arxiv_id":"1901.00137","n_code_links":0,"syntology":null},{"paper":"/paper/generative-adversarial-user-model-for","slug":"generative-adversarial-user-model-for","title":"Generative Adversarial User Model for Reinforcement Learning Based Recommendation System","date":"2018-12-27","arxiv_id":"1812.10613","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["xinshi-chen/GenerativeAdversarialUserModel"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"parallelized-interactive-machine-learning-on","title":"Parallelized Interactive Machine Learning on Autonomous Vehicles","date":"2018-12-23","arxiv_id":"1812.09724","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-navigate-the-web","title":"Learning to Navigate the Web","date":"2018-12-21","arxiv_id":"1812.09195","n_code_links":0,"syntology":null},{"paper":null,"slug":"double-deep-q-learning-for-optimal-execution","title":"Double Deep Q-Learning for Optimal Execution","date":"2018-12-17","arxiv_id":"1812.06600","n_code_links":0,"syntology":null},{"paper":"/paper/decentralized-computation-offloading-for","slug":"decentralized-computation-offloading-for","title":"Decentralized Computation Offloading for Multi-User Mobile Edge Computing: A Deep Reinforcement Learning Approach","date":"2018-12-16","arxiv_id":"1812.07394","n_code_links":2,"syntology":null},{"paper":"/paper/off-policy-deep-reinforcement-learning","slug":"off-policy-deep-reinforcement-learning","title":"Off-Policy Deep Reinforcement Learning without Exploration","date":"2018-12-07","arxiv_id":"1812.02900","n_code_links":10,"syntology":{"ran":14,"of":14,"n_ran_checked":14,"n_instrument":0,"unverified":0,"pointer_only":9,"phrase":"14 ran (of which 12 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sfujim/BCQ"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"active-deep-q-learning-with-demonstration","title":"Active Deep Q-learning with Demonstration","date":"2018-12-06","arxiv_id":"1812.02632","n_code_links":0,"syntology":null},{"paper":null,"slug":"bach2bach-generating-music-using-a-deep","title":"Bach2Bach: Generating Music Using A Deep Reinforcement Learning Approach","date":"2018-12-03","arxiv_id":"1812.01060","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-intelligent","title":"Deep Reinforcement Learning for Intelligent Transportation Systems","date":"2018-12-03","arxiv_id":"1812.00979","n_code_links":0,"syntology":null},{"paper":"/paper/macro-action-selection-with-deep","slug":"macro-action-selection-with-deep","title":"Macro action selection with deep reinforcement learning in StarCraft","date":"2018-12-02","arxiv_id":"1812.00336","n_code_links":1,"syntology":null},{"paper":"/paper/deep-multi-agent-reinforcement-learning-with","slug":"deep-multi-agent-reinforcement-learning-with","title":"Deep Multi-Agent Reinforcement Learning with Relevance Graphs","date":"2018-11-30","arxiv_id":"1811.12557","n_code_links":1,"syntology":{"ran":1,"of":7,"n_ran_checked":1,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["tegg89/magnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/urban-driving-with-multi-objective-deep","slug":"urban-driving-with-multi-objective-deep","title":"Urban Driving with Multi-Objective Deep Reinforcement Learning","date":"2018-11-21","arxiv_id":"1811.08586","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-initial-attempt-of-combining-visual","title":"An initial attempt of combining visual selective attention with deep reinforcement learning","date":"2018-11-11","arxiv_id":"1811.04407","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-eligibility-traces-for-deep","slug":"efficient-eligibility-traces-for-deep","title":"Reconciling $λ$-Returns with Experience Replay","date":"2018-10-23","arxiv_id":"1810.09967","n_code_links":1,"syntology":null},{"paper":"/paper/successor-uncertainties-exploration-and","slug":"successor-uncertainties-exploration-and","title":"Successor Uncertainties: Exploration and Uncertainty in Temporal Difference Learning","date":"2018-10-15","arxiv_id":"1810.06530","n_code_links":2,"syntology":{"ran":7,"of":7,"n_ran_checked":2,"n_instrument":5,"unverified":0,"pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/empowerment-driven-exploration-using-mutual","slug":"empowerment-driven-exploration-using-mutual","title":"Empowerment-driven Exploration using Mutual Information Estimation","date":"2018-10-11","arxiv_id":"1810.05533","n_code_links":1,"syntology":null},{"paper":"/paper/parametrized-deep-q-networks-learning","slug":"parametrized-deep-q-networks-learning","title":"Parametrized Deep Q-Networks Learning: Reinforcement Learning with Discrete-Continuous Hybrid Action Space","date":"2018-10-10","arxiv_id":"1810.06394","n_code_links":5,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/generalization-and-regularization-in-dqn","slug":"generalization-and-regularization-in-dqn","title":"Generalization and Regularization in DQN","date":"2018-09-29","arxiv_id":"1810.00123","n_code_links":1,"syntology":null},{"paper":null,"slug":"coordinated-heterogeneous-distributed","title":"Coordinated Heterogeneous Distributed Perception based on Latent Space Representation","date":"2018-09-12","arxiv_id":"1809.04558","n_code_links":0,"syntology":null},{"paper":null,"slug":"learn-what-not-to-learn-action-elimination","title":"Learn What Not to Learn: Action Elimination with Deep Reinforcement Learning","date":"2018-09-06","arxiv_id":"1809.02121","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-based-regularization-for-deep","title":"Model-Based Regularization for Deep Reinforcement Learning with Transcoder Networks","date":"2018-09-06","arxiv_id":"1809.01906","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-using-augmented-neural","title":"Reinforcement Learning using Augmented Neural Networks","date":"2018-06-20","arxiv_id":"1806.07692","n_code_links":0,"syntology":null},{"paper":"/paper/surprising-negative-results-for-generative","slug":"surprising-negative-results-for-generative","title":"Surprising Negative Results for Generative Adversarial Tree Search","date":"2018-06-15","arxiv_id":"1806.05780","n_code_links":3,"syntology":null},{"paper":"/paper/implicit-quantile-networks-for-distributional","slug":"implicit-quantile-networks-for-distributional","title":"Implicit Quantile Networks for Distributional Reinforcement Learning","date":"2018-06-14","arxiv_id":"1806.06923","n_code_links":19,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"qualitative-measurements-of-policy","title":"Qualitative Measurements of Policy Discrepancy for Return-Based Deep Q-Network","date":"2018-06-14","arxiv_id":"1806.06953","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-search-in-long-documents-using","slug":"learning-to-search-in-long-documents-using","title":"Learning to Search in Long Documents Using Document Structure","date":"2018-06-09","arxiv_id":"1806.03529","n_code_links":1,"syntology":null},{"paper":"/paper/randomized-value-functions-via-multiplicative","slug":"randomized-value-functions-via-multiplicative","title":"Randomized Value Functions via Multiplicative Normalizing Flows","date":"2018-06-06","arxiv_id":"1806.02315","n_code_links":2,"syntology":null},{"paper":"/paper/sample-efficient-deep-reinforcement-learning-2","slug":"sample-efficient-deep-reinforcement-learning-2","title":"Sample-Efficient Deep Reinforcement Learning via Episodic Backward Update","date":"2018-05-31","arxiv_id":"1805.12375","n_code_links":1,"syntology":null},{"paper":null,"slug":"episodic-memory-deep-q-networks","title":"Episodic Memory Deep Q-Networks","date":"2018-05-19","arxiv_id":"1805.07603","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimized-computation-offloading-performance","title":"Optimized Computation Offloading Performance in Virtual Edge Computing Systems via Deep Reinforcement Learning","date":"2018-05-16","arxiv_id":"1805.06146","n_code_links":0,"syntology":null},{"paper":"/paper/advances-in-experience-replay","slug":"advances-in-experience-replay","title":"Advances in Experience Replay","date":"2018-05-15","arxiv_id":"1805.05536","n_code_links":1,"syntology":null},{"paper":null,"slug":"movi-a-model-free-approach-to-dynamic-fleet","title":"MOVI: A Model-Free Approach to Dynamic Fleet Management","date":"2018-04-13","arxiv_id":"1804.04758","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-qosqoe-aware","title":"Reinforcement Learning based QoS/QoE-aware Service Function Chaining in Software-Driven 5G Slices","date":"2018-04-06","arxiv_id":"1804.02099","n_code_links":0,"syntology":null},{"paper":null,"slug":"natural-gradient-deep-q-learning","title":"Natural Gradient Deep Q-learning","date":"2018-03-20","arxiv_id":"1803.07482","n_code_links":0,"syntology":null},{"paper":null,"slug":"weighted-double-deep-multiagent-reinforcement","title":"Weighted Double Deep Multiagent Reinforcement Learning in Stochastic Cooperative Environments","date":"2018-02-23","arxiv_id":"1802.08534","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-exploration-through-bayesian-deep-q","slug":"efficient-exploration-through-bayesian-deep-q","title":"Efficient Exploration through Bayesian Deep Q-Networks","date":"2018-02-13","arxiv_id":"1802.04412","n_code_links":1,"syntology":null},{"paper":null,"slug":"faster-deep-q-learning-using-neural-episodic","title":"Faster Deep Q-learning using Neural Episodic Control","date":"2018-01-06","arxiv_id":"1801.01968","n_code_links":0,"syntology":null},{"paper":null,"slug":"autonomous-vehicle-fleet-coordination-with","title":"Autonomous Vehicle Fleet Coordination With Deep Reinforcement Learning","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"faster-reinforcement-learning-with-expert","title":"Faster Reinforcement Learning with Expert State Sequences","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/parametrized-deep-q-networks-learning-playing","slug":"parametrized-deep-q-networks-learning-playing","title":"PARAMETRIZED DEEP Q-NETWORKS LEARNING: PLAYING ONLINE BATTLE ARENA WITH DISCRETE-CONTINUOUS HYBRID ACTION SPACE","date":"2018-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"a-deep-policy-inference-q-network-for-multi","title":"A Deep Policy Inference Q-Network for Multi-Agent Systems","date":"2017-12-21","arxiv_id":"1712.07893","n_code_links":0,"syntology":null},{"paper":"/paper/deep-neuroevolution-genetic-algorithms-are-a","slug":"deep-neuroevolution-genetic-algorithms-are-a","title":"Deep Neuroevolution: Genetic Algorithms Are a Competitive Alternative for Training Deep Neural Networks for Reinforcement Learning","date":"2017-12-18","arxiv_id":"1712.06567","n_code_links":12,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"uncertainty-estimates-for-efficient-neural","title":"Uncertainty Estimates for Efficient Neural Network-based Dialogue Policy Optimisation","date":"2017-11-30","arxiv_id":"1711.11486","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-benchmarking-environment-for-reinforcement","title":"A Benchmarking Environment for Reinforcement Learning Based Task Oriented Dialogue Management","date":"2017-11-29","arxiv_id":"1711.11023","n_code_links":0,"syntology":null},{"paper":"/paper/implementing-the-deep-q-network","slug":"implementing-the-deep-q-network","title":"Implementing the Deep Q-Network","date":"2017-11-20","arxiv_id":"1711.07478","n_code_links":1,"syntology":null},{"paper":"/paper/treeqn-and-atreec-differentiable-tree","slug":"treeqn-and-atreec-differentiable-tree","title":"TreeQN and ATreeC: Differentiable Tree-Structured Models for Deep Reinforcement Learning","date":"2017-10-31","arxiv_id":"1710.11417","n_code_links":1,"syntology":{"ran":7,"of":12,"n_ran_checked":7,"n_instrument":0,"unverified":5,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["oxwhirl/treeqn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/distributional-reinforcement-learning-with-1","slug":"distributional-reinforcement-learning-with-1","title":"Distributional Reinforcement Learning with Quantile Regression","date":"2017-10-27","arxiv_id":"1710.10044","n_code_links":17,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/rainbow-combining-improvements-in-deep","slug":"rainbow-combining-improvements-in-deep","title":"Rainbow: Combining Improvements in Deep Reinforcement Learning","date":"2017-10-06","arxiv_id":"1710.02298","n_code_links":34,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"deep-reinforcement-learning-with-surrogate","title":"Deep Reinforcement Learning with Surrogate Agent-Environment Interface","date":"2017-09-12","arxiv_id":"1709.03942","n_code_links":0,"syntology":null},{"paper":null,"slug":"pre-training-neural-networks-with-human","title":"Pre-training Neural Networks with Human Demonstrations for Deep Reinforcement Learning","date":"2017-09-12","arxiv_id":"1709.04083","n_code_links":0,"syntology":null},{"paper":null,"slug":"formulation-of-deep-reinforcement-learning","title":"Formulation of Deep Reinforcement Learning Architecture Toward Autonomous Driving for On-Ramp Merge","date":"2017-09-07","arxiv_id":"1709.02066","n_code_links":0,"syntology":null},{"paper":null,"slug":"ladder-a-human-level-bidding-agent-for-large","title":"LADDER: A Human-Level Bidding Agent for Large-Scale Real-Time Online Auctions","date":"2017-08-18","arxiv_id":"1708.05565","n_code_links":0,"syntology":null},{"paper":null,"slug":"3dcnn-dqn-rnn-a-deep-reinforcement-learning","title":"3DCNN-DQN-RNN: A Deep Reinforcement Learning Framework for Semantic Parsing of Large-scale 3D Point Clouds","date":"2017-07-21","arxiv_id":"1707.06783","n_code_links":0,"syntology":null},{"paper":"/paper/noisy-networks-for-exploration","slug":"noisy-networks-for-exploration","title":"Noisy Networks for Exploration","date":"2017-06-30","arxiv_id":"1706.10295","n_code_links":15,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/parameter-space-noise-for-exploration","slug":"parameter-space-noise-for-exploration","title":"Parameter Space Noise for Exploration","date":"2017-06-06","arxiv_id":"1706.01905","n_code_links":10,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"explaining-transition-systems-through-program","title":"Explaining Transition Systems through Program Induction","date":"2017-05-23","arxiv_id":"1705.08320","n_code_links":0,"syntology":null},{"paper":null,"slug":"shallow-updates-for-deep-reinforcement","title":"Shallow Updates for Deep Reinforcement Learning","date":"2017-05-21","arxiv_id":"1705.07461","n_code_links":0,"syntology":null},{"paper":"/paper/the-reactor-a-fast-and-sample-efficient-actor","slug":"the-reactor-a-fast-and-sample-efficient-actor","title":"The Reactor: A fast and sample-efficient Actor-Critic agent for Reinforcement Learning","date":"2017-04-15","arxiv_id":"1704.04651","n_code_links":0,"syntology":null},{"paper":"/paper/deep-q-learning-from-demonstrations","slug":"deep-q-learning-from-demonstrations","title":"Deep Q-learning from Demonstrations","date":"2017-04-12","arxiv_id":"1704.03732","n_code_links":6,"syntology":null},{"paper":null,"slug":"tactics-of-adversarial-attack-on-deep","title":"Tactics of Adversarial Attack on Deep Reinforcement Learning Agents","date":"2017-03-08","arxiv_id":"1703.06748","n_code_links":0,"syntology":null},{"paper":"/paper/count-based-exploration-with-neural-density","slug":"count-based-exploration-with-neural-density","title":"Count-Based Exploration with Neural Density Models","date":"2017-03-03","arxiv_id":"1703.01310","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/sigmoid-weighted-linear-units-for-neural","slug":"sigmoid-weighted-linear-units-for-neural","title":"Sigmoid-Weighted Linear Units for Neural Network Function Approximation in Reinforcement Learning","date":"2017-02-10","arxiv_id":"1702.03118","n_code_links":0,"syntology":null},{"paper":"/paper/autonomous-braking-system-via-deep","slug":"autonomous-braking-system-via-deep","title":"Autonomous Braking System via Deep Reinforcement Learning","date":"2017-02-08","arxiv_id":"1702.02302","n_code_links":2,"syntology":null},{"paper":"/paper/vulnerability-of-deep-reinforcement-learning","slug":"vulnerability-of-deep-reinforcement-learning","title":"Vulnerability of Deep Reinforcement Learning to Policy Induction Attacks","date":"2017-01-16","arxiv_id":"1701.04143","n_code_links":1,"syntology":null}],"record_sha256":"7f6c2148b3ae6547cbf22f430d3500ddea4ef6746ce61a2b458b943380d0d9d7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}