{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/16","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":16,"pages_in_order":18,"rows_per_page":100,"rows":[1501,1600],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning/papers/15","next":"/method/q-learning/papers/17","papers":[{"paper":null,"slug":"reinforcement-learning-of-markov-decision","title":"Reinforcement Learning of Markov Decision Processes with Peak Constraints","date":"2019-01-23","arxiv_id":"1901.07839","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-multi-step-deep-reinforcement","slug":"understanding-multi-step-deep-reinforcement","title":"Understanding Multi-Step Deep Reinforcement Learning: A Systematic Study of the DQN Target","date":"2019-01-22","arxiv_id":"1901.07510","n_code_links":1,"syntology":null},{"paper":"/paper/a-deep-recurrent-q-network-towards-self","slug":"a-deep-recurrent-q-network-towards-self","title":"A Deep Recurrent Q Network towards Self-adapting Distributed Microservices architecture","date":"2019-01-13","arxiv_id":"1901.04011","n_code_links":1,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-for-imbalanced","slug":"deep-reinforcement-learning-for-imbalanced","title":"Deep Reinforcement Learning for Imbalanced Classification","date":"2019-01-05","arxiv_id":"1901.01379","n_code_links":3,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["linenus/DRL-For-imbalanced-Classification"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"accelerating-goal-directed-reinforcement","title":"Accelerating Goal-Directed Reinforcement Learning by Model Characterization","date":"2019-01-04","arxiv_id":"1901.01977","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-theoretical-analysis-of-deep-q-learning","title":"A Theoretical Analysis of Deep Q-Learning","date":"2019-01-01","arxiv_id":"1901.00137","n_code_links":0,"syntology":null},{"paper":"/paper/generative-adversarial-user-model-for","slug":"generative-adversarial-user-model-for","title":"Generative Adversarial User Model for Reinforcement Learning Based Recommendation System","date":"2018-12-27","arxiv_id":"1812.10613","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["xinshi-chen/GenerativeAdversarialUserModel"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"parallelized-interactive-machine-learning-on","title":"Parallelized Interactive Machine Learning on Autonomous Vehicles","date":"2018-12-23","arxiv_id":"1812.09724","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-navigate-the-web","title":"Learning to Navigate the Web","date":"2018-12-21","arxiv_id":"1812.09195","n_code_links":0,"syntology":null},{"paper":null,"slug":"double-deep-q-learning-for-optimal-execution","title":"Double Deep Q-Learning for Optimal Execution","date":"2018-12-17","arxiv_id":"1812.06600","n_code_links":0,"syntology":null},{"paper":"/paper/decentralized-computation-offloading-for","slug":"decentralized-computation-offloading-for","title":"Decentralized Computation Offloading for Multi-User Mobile Edge Computing: A Deep Reinforcement Learning Approach","date":"2018-12-16","arxiv_id":"1812.07394","n_code_links":2,"syntology":null},{"paper":null,"slug":"learning-sharing-behaviors-with-arbitrary","title":"Learning Sharing Behaviors with Arbitrary Numbers of Agents","date":"2018-12-10","arxiv_id":"1812.04145","n_code_links":0,"syntology":null},{"paper":"/paper/off-policy-deep-reinforcement-learning","slug":"off-policy-deep-reinforcement-learning","title":"Off-Policy Deep Reinforcement Learning without Exploration","date":"2018-12-07","arxiv_id":"1812.02900","n_code_links":10,"syntology":{"ran":14,"of":14,"n_ran_checked":14,"n_instrument":0,"unverified":0,"pointer_only":9,"phrase":"14 ran (of which 12 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sfujim/BCQ"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"active-deep-q-learning-with-demonstration","title":"Active Deep Q-learning with Demonstration","date":"2018-12-06","arxiv_id":"1812.02632","n_code_links":0,"syntology":null},{"paper":null,"slug":"bach2bach-generating-music-using-a-deep","title":"Bach2Bach: Generating Music Using A Deep Reinforcement Learning Approach","date":"2018-12-03","arxiv_id":"1812.01060","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-intelligent","title":"Deep Reinforcement Learning for Intelligent Transportation Systems","date":"2018-12-03","arxiv_id":"1812.00979","n_code_links":0,"syntology":null},{"paper":"/paper/macro-action-selection-with-deep","slug":"macro-action-selection-with-deep","title":"Macro action selection with deep reinforcement learning in StarCraft","date":"2018-12-02","arxiv_id":"1812.00336","n_code_links":1,"syntology":null},{"paper":"/paper/revisiting-the-softmax-bellman-operator","slug":"revisiting-the-softmax-bellman-operator","title":"Revisiting the Softmax Bellman Operator: New Benefits and New Perspective","date":"2018-12-02","arxiv_id":"1812.00456","n_code_links":2,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["zhao-song/Softmax-DQN"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"non-delusional-q-learning-and-value-iteration","title":"Non-delusional Q-learning and value-iteration","date":"2018-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/deep-multi-agent-reinforcement-learning-with","slug":"deep-multi-agent-reinforcement-learning-with","title":"Deep Multi-Agent Reinforcement Learning with Relevance Graphs","date":"2018-11-30","arxiv_id":"1811.12557","n_code_links":1,"syntology":{"ran":1,"of":7,"n_ran_checked":1,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["tegg89/magnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/urban-driving-with-multi-objective-deep","slug":"urban-driving-with-multi-objective-deep","title":"Urban Driving with Multi-Objective Deep Reinforcement Learning","date":"2018-11-21","arxiv_id":"1811.08586","n_code_links":1,"syntology":null},{"paper":"/paper/reinforcement-learning-with-a-and-a-deep","slug":"reinforcement-learning-with-a-and-a-deep","title":"Reinforcement Learning with A* and a Deep Heuristic","date":"2018-11-19","arxiv_id":"1811.07745","n_code_links":2,"syntology":null},{"paper":null,"slug":"emergence-of-addictive-behaviors-in","title":"Emergence of Addictive Behaviors in Reinforcement Learning Agents","date":"2018-11-14","arxiv_id":"1811.05590","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-initial-attempt-of-combining-visual","title":"An initial attempt of combining visual selective attention with deep reinforcement learning","date":"2018-11-11","arxiv_id":"1811.04407","n_code_links":0,"syntology":null},{"paper":null,"slug":"managing-app-install-ad-campaigns-in-rtb-a-q","title":"Managing App Install Ad Campaigns in RTB: A Q-Learning Approach","date":"2018-11-11","arxiv_id":"1811.04475","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-green","title":"Deep Reinforcement Learning for Green Security Games with Real-Time Information","date":"2018-11-06","arxiv_id":"1811.02483","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-dynamic-model","title":"Reinforcement Learning based Dynamic Model Selection for Short-Term Load Forecasting","date":"2018-11-05","arxiv_id":"1811.01846","n_code_links":0,"syntology":null},{"paper":null,"slug":"approximate-dynamic-oracle-for-dependency","title":"Approximate Dynamic Oracle for Dependency Parsing with Reinforcement Learning","date":"2018-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"structure-learning-of-deep-neural-networks","title":"Structure Learning of Deep Neural Networks with Q-Learning","date":"2018-10-31","arxiv_id":"1810.13155","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributive-dynamic-spectrum-access-through","title":"Distributive Dynamic Spectrum Access through Deep Reinforcement Learning: A Reservoir Computing Based Approach","date":"2018-10-28","arxiv_id":"1810.11758","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-negotiating-behavior-between-cars-in","title":"Learning Negotiating Behavior Between Cars in Intersections using Deep Q-Learning","date":"2018-10-24","arxiv_id":"1810.10469","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-eligibility-traces-for-deep","slug":"efficient-eligibility-traces-for-deep","title":"Reconciling $λ$-Returns with Experience Replay","date":"2018-10-23","arxiv_id":"1810.09967","n_code_links":1,"syntology":null},{"paper":"/paper/actor-expert-a-framework-for-using-action","slug":"actor-expert-a-framework-for-using-action","title":"Greedy Actor-Critic: A New Conditional Cross-Entropy Method for Policy Improvement","date":"2018-10-22","arxiv_id":"1810.09103","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["samuelfneumann/greedyac"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/successor-uncertainties-exploration-and","slug":"successor-uncertainties-exploration-and","title":"Successor Uncertainties: Exploration and Uncertainty in Temporal Difference Learning","date":"2018-10-15","arxiv_id":"1810.06530","n_code_links":2,"syntology":{"ran":7,"of":7,"n_ran_checked":2,"n_instrument":5,"unverified":0,"pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/assessing-the-potential-of-classical-q","slug":"assessing-the-potential-of-classical-q","title":"Assessing the Potential of Classical Q-learning in General Game Playing","date":"2018-10-14","arxiv_id":"1810.06078","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-sketch-with-deep-q-networks-and","title":"Learning to Sketch with Deep Q Networks and Demonstrated Strokes","date":"2018-10-14","arxiv_id":"1810.05977","n_code_links":0,"syntology":null},{"paper":"/paper/empowerment-driven-exploration-using-mutual","slug":"empowerment-driven-exploration-using-mutual","title":"Empowerment-driven Exploration using Mutual Information Estimation","date":"2018-10-11","arxiv_id":"1810.05533","n_code_links":1,"syntology":null},{"paper":"/paper/parametrized-deep-q-networks-learning","slug":"parametrized-deep-q-networks-learning","title":"Parametrized Deep Q-Networks Learning: Reinforcement Learning with Discrete-Continuous Hybrid Action Space","date":"2018-10-10","arxiv_id":"1810.06394","n_code_links":5,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/deep-quality-value-dqv-learning","slug":"deep-quality-value-dqv-learning","title":"Deep Quality-Value (DQV) Learning","date":"2018-09-30","arxiv_id":"1810.00368","n_code_links":3,"syntology":null},{"paper":"/paper/generalization-and-regularization-in-dqn","slug":"generalization-and-regularization-in-dqn","title":"Generalization and Regularization in DQN","date":"2018-09-29","arxiv_id":"1810.00123","n_code_links":1,"syntology":null},{"paper":null,"slug":"target-transfer-q-learning-and-its","title":"Target Transfer Q-Learning and Its Convergence Analysis","date":"2018-09-21","arxiv_id":"1809.08923","n_code_links":0,"syntology":null},{"paper":null,"slug":"hidden-markov-model-estimation-based-q","title":"Hidden Markov Model Estimation-Based Q-learning for Partially Observable Markov Decision Process","date":"2018-09-17","arxiv_id":"1809.06401","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimal-matrix-momentum-stochastic","title":"Optimal Matrix Momentum Stochastic Approximation and Applications to Q-learning","date":"2018-09-17","arxiv_id":"1809.06277","n_code_links":0,"syntology":null},{"paper":"/paper/deterministic-implementations-for","slug":"deterministic-implementations-for","title":"Deterministic Implementations for Reproducibility in Deep Reinforcement Learning","date":"2018-09-15","arxiv_id":"1809.05676","n_code_links":1,"syntology":null},{"paper":"/paper/sampled-policy-gradient-for-learning-to-play","slug":"sampled-policy-gradient-for-learning-to-play","title":"Sampled Policy Gradient for Learning to Play the Game Agar.io","date":"2018-09-15","arxiv_id":"1809.05763","n_code_links":2,"syntology":null},{"paper":"/paper/towards-better-interpretability-in-deep-q","slug":"towards-better-interpretability-in-deep-q","title":"Towards Better Interpretability in Deep Q-Networks","date":"2018-09-15","arxiv_id":"1809.05630","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/negative-update-intervals-in-deep-multi-agent","slug":"negative-update-intervals-in-deep-multi-agent","title":"Negative Update Intervals in Deep Multi-Agent Reinforcement Learning","date":"2018-09-13","arxiv_id":"1809.05096","n_code_links":1,"syntology":null},{"paper":null,"slug":"coordinated-heterogeneous-distributed","title":"Coordinated Heterogeneous Distributed Perception based on Latent Space Representation","date":"2018-09-12","arxiv_id":"1809.04558","n_code_links":0,"syntology":null},{"paper":null,"slug":"learn-what-not-to-learn-action-elimination","title":"Learn What Not to Learn: Action Elimination with Deep Reinforcement Learning","date":"2018-09-06","arxiv_id":"1809.02121","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-based-regularization-for-deep","title":"Model-Based Regularization for Deep Reinforcement Learning with Transcoder Networks","date":"2018-09-06","arxiv_id":"1809.01906","n_code_links":0,"syntology":null},{"paper":null,"slug":"directed-exploration-in-pac-model-free","title":"Directed Exploration in PAC Model-Free Reinforcement Learning","date":"2018-08-31","arxiv_id":"1808.10552","n_code_links":0,"syntology":null},{"paper":null,"slug":"marl-fwc-optimal-coordination-of-freeway","title":"MARL-FWC: Optimal Coordination of Freeway Traffic Control Measures","date":"2018-08-27","arxiv_id":"1808.09806","n_code_links":0,"syntology":null},{"paper":"/paper/blockqnn-efficient-block-wise-neural-network","slug":"blockqnn-efficient-block-wise-neural-network","title":"BlockQNN: Efficient Block-wise Neural Network Architecture Generation","date":"2018-08-16","arxiv_id":"1808.05584","n_code_links":2,"syntology":null},{"paper":null,"slug":"automatic-derivation-of-formulas-using","title":"Automatic Derivation Of Formulas Using Reforcement Learning","date":"2018-08-15","arxiv_id":"1808.04946","n_code_links":0,"syntology":null},{"paper":"/paper/a-framework-for-automated-cellular-network","slug":"a-framework-for-automated-cellular-network","title":"A Framework for Automated Cellular Network Tuning with Reinforcement Learning","date":"2018-08-13","arxiv_id":"1808.05140","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-reinforcement-learning-approach-to-target","title":"A Reinforcement Learning Approach to Target Tracking in a Camera Network","date":"2018-07-26","arxiv_id":"1807.10336","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerated-structure-aware-reinforcement","title":"Accelerated Structure-Aware Reinforcement Learning for Delay-Sensitive Energy Harvesting Wireless Sensors","date":"2018-07-22","arxiv_id":"1807.08315","n_code_links":0,"syntology":null},{"paper":null,"slug":"discrete-linear-complexity-reinforcement","title":"Discrete linear-complexity reinforcement learning in continuous action spaces for Q-learning algorithms","date":"2018-07-16","arxiv_id":"1807.06957","n_code_links":0,"syntology":null},{"paper":"/paper/is-q-learning-provably-efficient","slug":"is-q-learning-provably-efficient","title":"Is Q-learning Provably Efficient?","date":"2018-07-10","arxiv_id":"1807.03765","n_code_links":1,"syntology":null},{"paper":null,"slug":"video-summarisation-by-classification-with","title":"Video Summarisation by Classification with Deep Reinforcement Learning","date":"2018-07-09","arxiv_id":"1807.03089","n_code_links":0,"syntology":null},{"paper":null,"slug":"playing-against-nature-causal-discovery-for","title":"Playing against Nature: causal discovery for decision making under uncertainty","date":"2018-07-03","arxiv_id":"1807.01268","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-explore-via-meta-policy-gradient","title":"Learning to Explore via Meta-Policy Gradient","date":"2018-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/using-reward-machines-for-high-level-task","slug":"using-reward-machines-for-high-level-task","title":"Using Reward Machines for High-Level Task Specification and Decomposition in Reinforcement Learning","date":"2018-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"many-goals-reinforcement-learning","title":"Many-Goals Reinforcement Learning","date":"2018-06-22","arxiv_id":"1806.09605","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-using-augmented-neural","title":"Reinforcement Learning using Augmented Neural Networks","date":"2018-06-20","arxiv_id":"1806.07692","n_code_links":0,"syntology":null},{"paper":"/paper/surprising-negative-results-for-generative","slug":"surprising-negative-results-for-generative","title":"Surprising Negative Results for Generative Adversarial Tree Search","date":"2018-06-15","arxiv_id":"1806.05780","n_code_links":3,"syntology":null},{"paper":"/paper/implicit-quantile-networks-for-distributional","slug":"implicit-quantile-networks-for-distributional","title":"Implicit Quantile Networks for Distributional Reinforcement Learning","date":"2018-06-14","arxiv_id":"1806.06923","n_code_links":19,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"qualitative-measurements-of-policy","title":"Qualitative Measurements of Policy Discrepancy for Return-Based Deep Q-Network","date":"2018-06-14","arxiv_id":"1806.06953","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-formation-of-the-structure-of","title":"Automatic formation of the structure of abstract machines in hierarchical reinforcement learning with state clustering","date":"2018-06-13","arxiv_id":"1806.05292","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-search-in-long-documents-using","slug":"learning-to-search-in-long-documents-using","title":"Learning to Search in Long Documents Using Document Structure","date":"2018-06-09","arxiv_id":"1806.03529","n_code_links":1,"syntology":null},{"paper":null,"slug":"fidelity-based-probabilistic-q-learning-for","title":"Fidelity-based Probabilistic Q-learning for Control of Quantum Systems","date":"2018-06-08","arxiv_id":"1806.03145","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-finite-time-analysis-of-temporal-difference","title":"A Finite Time Analysis of Temporal Difference Learning With Linear Function Approximation","date":"2018-06-06","arxiv_id":"1806.02450","n_code_links":0,"syntology":null},{"paper":"/paper/randomized-value-functions-via-multiplicative","slug":"randomized-value-functions-via-multiplicative","title":"Randomized Value Functions via Multiplicative Normalizing Flows","date":"2018-06-06","arxiv_id":"1806.02315","n_code_links":2,"syntology":null},{"paper":null,"slug":"hyperparameter-optimization-for-tracking-with","title":"Hyperparameter Optimization for Tracking With Continuous Deep Q-Learning","date":"2018-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/sample-efficient-deep-reinforcement-learning-2","slug":"sample-efficient-deep-reinforcement-learning-2","title":"Sample-Efficient Deep Reinforcement Learning via Episodic Backward Update","date":"2018-05-31","arxiv_id":"1805.12375","n_code_links":1,"syntology":null},{"paper":null,"slug":"depth-and-nonlinearity-induce-implicit","title":"Depth and nonlinearity induce implicit exploration for RL","date":"2018-05-29","arxiv_id":"1805.11711","n_code_links":0,"syntology":null},{"paper":null,"slug":"episodic-memory-deep-q-networks","title":"Episodic Memory Deep Q-Networks","date":"2018-05-19","arxiv_id":"1805.07603","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimized-computation-offloading-performance","title":"Optimized Computation Offloading Performance in Virtual Edge Computing Systems via Deep Reinforcement Learning","date":"2018-05-16","arxiv_id":"1805.06146","n_code_links":0,"syntology":null},{"paper":"/paper/advances-in-experience-replay","slug":"advances-in-experience-replay","title":"Advances in Experience Replay","date":"2018-05-15","arxiv_id":"1805.05536","n_code_links":1,"syntology":null},{"paper":null,"slug":"planning-and-learning-with-stochastic-action","title":"Planning and Learning with Stochastic Action Sets","date":"2018-05-07","arxiv_id":"1805.02363","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-hybrid-q-learning-sine-cosine-based","title":"A Hybrid Q-Learning Sine-Cosine-based Strategy for Addressing the Combinatorial Test Suite Minimization Problem","date":"2018-04-27","arxiv_id":"1805.00873","n_code_links":0,"syntology":null},{"paper":null,"slug":"multiagent-soft-q-learning","title":"Multiagent Soft Q-Learning","date":"2018-04-25","arxiv_id":"1804.09817","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-projective-simulation-in","title":"Benchmarking projective simulation in navigation problems","date":"2018-04-23","arxiv_id":"1804.08607","n_code_links":0,"syntology":null},{"paper":"/paper/towards-symbolic-reinforcement-learning-with","slug":"towards-symbolic-reinforcement-learning-with","title":"Towards Symbolic Reinforcement Learning with Common Sense","date":"2018-04-23","arxiv_id":"1804.08597","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforced-co-training","title":"Reinforced Co-Training","date":"2018-04-17","arxiv_id":"1804.06035","n_code_links":0,"syntology":null},{"paper":null,"slug":"state-augmentation-transformations-for-risk","title":"State-Augmentation Transformations for Risk-Sensitive Reinforcement Learning","date":"2018-04-16","arxiv_id":"1804.05950","n_code_links":0,"syntology":null},{"paper":"/paper/cytonrl-an-efficient-reinforcement-learning","slug":"cytonrl-an-efficient-reinforcement-learning","title":"CytonRL: an Efficient Reinforcement Learning Open-source Toolkit Implemented in C++","date":"2018-04-14","arxiv_id":"1804.05834","n_code_links":1,"syntology":null},{"paper":null,"slug":"movi-a-model-free-approach-to-dynamic-fleet","title":"MOVI: A Model-Free Approach to Dynamic Fleet Management","date":"2018-04-13","arxiv_id":"1804.04758","n_code_links":0,"syntology":null},{"paper":null,"slug":"hierarchical-modular-reinforcement-learning","title":"Hierarchical Modular Reinforcement Learning Method and Knowledge Acquisition of State-Action Rule for Multi-target Problem","date":"2018-04-08","arxiv_id":"1804.02698","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-qosqoe-aware","title":"Reinforcement Learning based QoS/QoE-aware Service Function Chaining in Software-Driven 5G Slices","date":"2018-04-06","arxiv_id":"1804.02099","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-learning-of-interactive-spoken-content","title":"Joint Learning of Interactive Spoken Content Retrieval and Trainable User Simulator","date":"2018-04-01","arxiv_id":"1804.00318","n_code_links":0,"syntology":null},{"paper":"/paper/learning-synergies-between-pushing-and","slug":"learning-synergies-between-pushing-and","title":"Learning Synergies between Pushing and Grasping with Self-supervised Deep Reinforcement Learning","date":"2018-03-27","arxiv_id":"1803.09956","n_code_links":4,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["andyzeng/visual-pushing-grasping"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"natural-gradient-deep-q-learning","title":"Natural Gradient Deep Q-learning","date":"2018-03-20","arxiv_id":"1803.07482","n_code_links":0,"syntology":null},{"paper":"/paper/composable-deep-reinforcement-learning-for","slug":"composable-deep-reinforcement-learning-for","title":"Composable Deep Reinforcement Learning for Robotic Manipulation","date":"2018-03-19","arxiv_id":"1803.06773","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-explore-with-meta-policy-gradient","title":"Learning to Explore with Meta-Policy Gradient","date":"2018-03-13","arxiv_id":"1803.05044","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-for-time-series","slug":"deep-reinforcement-learning-for-time-series","title":"Deep reinforcement learning for time series: playing idealized trading games","date":"2018-03-11","arxiv_id":"1803.03916","n_code_links":2,"syntology":null},{"paper":null,"slug":"q-cp-learning-action-values-for-cooperative","title":"Q-CP: Learning Action Values for Cooperative Planning","date":"2018-03-01","arxiv_id":"1803.00297","n_code_links":0,"syntology":null},{"paper":null,"slug":"variance-reduction-methods-for-sublinear","title":"Variance Reduction Methods for Sublinear Reinforcement Learning","date":"2018-02-26","arxiv_id":"1802.09184","n_code_links":0,"syntology":null},{"paper":null,"slug":"weighted-double-deep-multiagent-reinforcement","title":"Weighted Double Deep Multiagent Reinforcement Learning in Stochastic Cooperative Environments","date":"2018-02-23","arxiv_id":"1802.08534","n_code_links":0,"syntology":null},{"paper":"/paper/a-deep-q-learning-agent-for-the-l-game-with","slug":"a-deep-q-learning-agent-for-the-l-game-with","title":"A Deep Q-Learning Agent for the L-Game with Variable Batch Training","date":"2018-02-17","arxiv_id":"1802.06225","n_code_links":1,"syntology":null}],"record_sha256":"6819a6bfa15dde6d47ca0342f5bc921f2babdfb93255db8c12ff0723efad2e53","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}