{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/12","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":12,"pages_in_order":18,"rows_per_page":100,"rows":[1101,1200],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning/papers/11","next":"/method/q-learning/papers/13","papers":[{"paper":null,"slug":"real-time-active-vision-for-a-humanoid-soccer","title":"Real-time Active Vision for a Humanoid Soccer Robot Using Deep Reinforcement Learning","date":"2020-11-27","arxiv_id":"2011.13851","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-joint-path-and","title":"Reinforcement Learning-based Joint Path and Energy Optimization of Cellular-Connected Unmanned Aerial Vehicles","date":"2020-11-27","arxiv_id":"2011.13744","n_code_links":0,"syntology":null},{"paper":null,"slug":"predictive-per-balancing-priority-and","title":"Predictive PER: Balancing Priority and Diversity towards Stable Deep Reinforcement Learning","date":"2020-11-26","arxiv_id":"2011.13093","n_code_links":0,"syntology":null},{"paper":null,"slug":"diluted-near-optimal-expert-demonstrations","title":"Diluted Near-Optimal Expert Demonstrations for Guiding Dialogue Stochastic Policy Optimisation","date":"2020-11-25","arxiv_id":"2012.04687","n_code_links":0,"syntology":null},{"paper":"/paper/learning-principle-of-least-action-with","slug":"learning-principle-of-least-action-with","title":"Learning Principle of Least Action with Reinforcement Learning","date":"2020-11-24","arxiv_id":"2011.11891","n_code_links":1,"syntology":null},{"paper":"/paper/solving-the-lunar-lander-problem-under","slug":"solving-the-lunar-lander-problem-under","title":"Solving The Lunar Lander Problem under Uncertainty using Reinforcement Learning","date":"2020-11-24","arxiv_id":"2011.11850","n_code_links":2,"syntology":null},{"paper":"/paper/revisiting-rainbow-promoting-more-insightful","slug":"revisiting-rainbow-promoting-more-insightful","title":"Revisiting Rainbow: Promoting more Insightful and Inclusive Deep Reinforcement Learning Research","date":"2020-11-20","arxiv_id":"2011.14826","n_code_links":2,"syntology":null},{"paper":"/paper/adaptive-contention-window-design-using-deep","slug":"adaptive-contention-window-design-using-deep","title":"Adaptive Contention Window Design using Deep Q-learning","date":"2020-11-18","arxiv_id":"2011.09418","n_code_links":1,"syntology":null},{"paper":null,"slug":"leveraging-the-variance-of-return-sequences-1","title":"Leveraging the Variance of Return Sequences for Exploration Policy","date":"2020-11-17","arxiv_id":"2011.08649","n_code_links":0,"syntology":null},{"paper":null,"slug":"constrained-model-free-reinforcement-learning","title":"Constrained Model-Free Reinforcement Learning for Process Optimization","date":"2020-11-16","arxiv_id":"2011.07925","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deep-q-learning-based-path-planning-and","title":"A deep Q-Learning based Path Planning and Navigation System for Firefighting Environments","date":"2020-11-12","arxiv_id":"2011.06450","n_code_links":0,"syntology":null},{"paper":"/paper/optimizing-large-scale-fleet-management-on-a","slug":"optimizing-large-scale-fleet-management-on-a","title":"Optimizing Large-Scale Fleet Management on a Road Network using Multi-Agent Deep Reinforcement Learning with Graph Neural Network","date":"2020-11-12","arxiv_id":"2011.06175","n_code_links":1,"syntology":null},{"paper":null,"slug":"hamiltonian-q-learning-leveraging-importance-1","title":"On Using Hamiltonian Monte Carlo Sampling for Reinforcement Learning Problems in High-dimension","date":"2020-11-11","arxiv_id":"2011.05927","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-agent-reinforcement-learning-for-joint","title":"Multi-Agent Reinforcement Learning for Channel Assignment and Power Allocation in Platoon-Based C-V2X Systems","date":"2020-11-09","arxiv_id":"2011.04555","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforced-deep-markov-models-with","title":"Reinforced Deep Markov Models With Applications in Automatic Trading","date":"2020-11-09","arxiv_id":"2011.04391","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-assignment-problem","title":"Reinforcement Learning for Assignment problem","date":"2020-11-08","arxiv_id":"2011.03909","n_code_links":0,"syntology":null},{"paper":"/paper/control-with-adaptive-q-learning","slug":"control-with-adaptive-q-learning","title":"Control with adaptive Q-learning","date":"2020-11-03","arxiv_id":"2011.02141","n_code_links":1,"syntology":null},{"paper":null,"slug":"deepfoldit-a-deep-reinforcement-learning","title":"DeepFoldit -- A Deep Reinforcement Learning Neural Network Folding Proteins","date":"2020-10-28","arxiv_id":"2011.03442","n_code_links":0,"syntology":null},{"paper":null,"slug":"finite-time-analysis-of-decentralized","title":"Finite-Time Convergence Rates of Decentralized Stochastic Approximation with Applications in Multi-Agent and Multi-Task Learning","date":"2020-10-28","arxiv_id":"2010.15088","n_code_links":0,"syntology":null},{"paper":null,"slug":"energy-consumption-and-battery-aging","title":"Energy Consumption and Battery Aging Minimization Using a Q-learning Strategy for a Battery/Ultracapacitor Electric Vehicle","date":"2020-10-27","arxiv_id":"2010.14115","n_code_links":0,"syntology":null},{"paper":"/paper/hamilton-jacobi-deep-q-learning-for","slug":"hamilton-jacobi-deep-q-learning-for","title":"Hamilton-Jacobi Deep Q-Learning for Deterministic Continuous-Time Systems with Lipschitz Continuous Controls","date":"2020-10-27","arxiv_id":"2010.14087","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-time-reduction-using-warm-start","title":"Learning Time Reduction Using Warm Start Methods for a Reinforcement Learning Based Supervisory Control in Hybrid Electric Vehicle Applications","date":"2020-10-27","arxiv_id":"2010.14575","n_code_links":0,"syntology":null},{"paper":null,"slug":"energy-and-service-priority-aware-trajectory","title":"Energy and Service-priority aware Trajectory Design for UAV-BSs using Double Q-Learning","date":"2020-10-26","arxiv_id":"2010.13346","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-reinforcement-learning-by-a-finite","title":"Enhancing reinforcement learning by a finite reward response filter with a case study in intelligent structural control","date":"2020-10-25","arxiv_id":"2010.15597","n_code_links":0,"syntology":null},{"paper":"/paper/learning-guidance-rewards-with-trajectory","slug":"learning-guidance-rewards-with-trajectory","title":"Learning Guidance Rewards with Trajectory-space Smoothing","date":"2020-10-23","arxiv_id":"2010.12718","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":1,"n_instrument":4,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tgangwani/GuidanceRewards"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"stabilizing-transformer-based-action-sequence","title":"Stabilizing Transformer-Based Action Sequence Generation For Q-Learning","date":"2020-10-23","arxiv_id":"2010.12698","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-attacks-on-deep-algorithmic","title":"Adversarial Attacks on Deep Algorithmic Trading Policies","date":"2020-10-22","arxiv_id":"2010.11388","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-graph-based-and-patient-demographics-aware","title":"A Weighted Heterogeneous Graph Based Dialogue System","date":"2020-10-21","arxiv_id":"2010.10699","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-surrogate-q-learning-for-autonomous","title":"Deep Surrogate Q-Learning for Autonomous Driving","date":"2020-10-21","arxiv_id":"2010.11278","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-information-asymmetry-in-competitive-multi","title":"On Information Asymmetry in Competitive Multi-Agent Reinforcement Learning: Convergence and Optimality","date":"2020-10-21","arxiv_id":"2010.10901","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-inference-with-multi-head-automata","title":"Language Inference with Multi-head Automata through Reinforcement Learning","date":"2020-10-20","arxiv_id":"2010.10141","n_code_links":0,"syntology":null},{"paper":null,"slug":"chance-constrained-control-with-lexicographic","title":"Chance-Constrained Control with Lexicographic Deep Reinforcement Learning","date":"2020-10-19","arxiv_id":"2010.09468","n_code_links":0,"syntology":null},{"paper":"/paper/connections-between-relational-event-model","slug":"connections-between-relational-event-model","title":"Connections between Relational Event Model and Inverse Reinforcement Learning for Characterizing Group Interaction Sequences","date":"2020-10-19","arxiv_id":"2010.09810","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-agent-reinforcement-learning-in-noma","title":"Multi-Agent Reinforcement Learning in NOMA-aided UAV Networks for Cellular Offloading","date":"2020-10-18","arxiv_id":"2010.09094","n_code_links":0,"syntology":null},{"paper":null,"slug":"noma-in-uav-aided-cellular-offloading-a","title":"NOMA in UAV-aided cellular offloading: A machine learning approach","date":"2020-10-18","arxiv_id":"2011.14776","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-dexterous-manipulation-from","title":"Learning Dexterous Manipulation from Suboptimal Experts","date":"2020-10-16","arxiv_id":"2010.08587","n_code_links":0,"syntology":null},{"paper":"/paper/multi-agent-collaboration-via-reward-1","slug":"multi-agent-collaboration-via-reward-1","title":"Multi-Agent Collaboration via Reward Attribution Decomposition","date":"2020-10-16","arxiv_id":"2010.08531","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-nesterov-s-accelerated-quasi-newton-method","title":"A Nesterov's Accelerated quasi-Newton method for Global Routing using Deep Reinforcement Learning","date":"2020-10-15","arxiv_id":"2010.09465","n_code_links":0,"syntology":null},{"paper":"/paper/epidemioptim-a-toolbox-for-the-optimization-1","slug":"epidemioptim-a-toolbox-for-the-optimization-1","title":"EpidemiOptim: A Toolbox for the Optimization of Control Policies in Epidemiological Models","date":"2020-10-09","arxiv_id":"2010.04452","n_code_links":2,"syntology":null},{"paper":"/paper/instance-weighted-incremental-evolution","slug":"instance-weighted-incremental-evolution","title":"Instance Weighted Incremental Evolution Strategies for Reinforcement Learning in Dynamic Environments","date":"2020-10-09","arxiv_id":"2010.04605","n_code_links":1,"syntology":null},{"paper":null,"slug":"parameterized-reinforcement-learning-for","title":"Parameterized Reinforcement Learning for Optical System Optimization","date":"2020-10-09","arxiv_id":"2010.05769","n_code_links":0,"syntology":null},{"paper":"/paper/q-learning-with-language-model-for-edit-based","slug":"q-learning-with-language-model-for-edit-based","title":"Q-learning with Language Model for Edit-based Unsupervised Summarization","date":"2020-10-09","arxiv_id":"2010.04379","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kohilin/ealm"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":null,"slug":"near-optimal-regret-bounds-for-model-free-rl-1","title":"Model-Free Non-Stationary RL: Near-Optimal Regret and Applications in Multi-Agent RL and Inventory Control","date":"2020-10-07","arxiv_id":"2010.03161","n_code_links":0,"syntology":null},{"paper":null,"slug":"machine-learning-empowered-trajectory-and","title":"Machine Learning Empowered Trajectory and Passive Beamforming Design in UAV-RIS Wireless Networks","date":"2020-10-06","arxiv_id":"2010.02749","n_code_links":0,"syntology":null},{"paper":null,"slug":"value-based-bayesian-meta-reinforcement","title":"Bayesian Meta-reinforcement Learning for Traffic Signal Control","date":"2020-10-01","arxiv_id":"2010.00163","n_code_links":0,"syntology":null},{"paper":null,"slug":"strategy-and-benchmark-for-converting-deep-q","title":"Strategy and Benchmark for Converting Deep Q-Networks to Event-Driven Spiking Neural Networks","date":"2020-09-30","arxiv_id":"2009.14456","n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-learning-in-deep-q-networks","title":"Cross Learning in Deep Q-Networks","date":"2020-09-29","arxiv_id":"2009.13780","n_code_links":0,"syntology":null},{"paper":null,"slug":"finite-time-analysis-for-double-q-learning","title":"Finite-Time Analysis for Double Q-learning","date":"2020-09-29","arxiv_id":"2009.14257","n_code_links":0,"syntology":null},{"paper":null,"slug":"lineage-evolution-reinforcement-learning","title":"Lineage Evolution Reinforcement Learning","date":"2020-09-26","arxiv_id":"2010.14616","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-approach-for-tactical-decision-making","title":"A New Approach for Tactical Decision Making in Lane Changing: Sample Efficient Deep Q Learning with a Safety Feedback Reward","date":"2020-09-24","arxiv_id":"2009.11905","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-q-learning-provably-efficient-an-extended","title":"Is Q-Learning Provably Efficient? An Extended Analysis","date":"2020-09-22","arxiv_id":"2009.10396","n_code_links":0,"syntology":null},{"paper":null,"slug":"hidden-incentives-for-auto-induced","title":"Hidden Incentives for Auto-Induced Distributional Shift","date":"2020-09-19","arxiv_id":"2009.09153","n_code_links":0,"syntology":null},{"paper":"/paper/energy-based-surprise-minimization-for-multi","slug":"energy-based-surprise-minimization-for-multi","title":"Energy-based Surprise Minimization for Multi-Agent Value Factorization","date":"2020-09-16","arxiv_id":"2009.09842","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-dynamic-resource","title":"Reinforcement Learning for Dynamic Resource Optimization in 5G Radio Access Network Slicing","date":"2020-09-14","arxiv_id":"2009.06579","n_code_links":0,"syntology":null},{"paper":null,"slug":"aoi-minimization-in-status-update-control","title":"AoI Minimization in Status Update Control with Energy Harvesting Sensors","date":"2020-09-09","arxiv_id":"2009.04224","n_code_links":0,"syntology":null},{"paper":null,"slug":"tactical-decision-making-for-emergency","title":"Tactical Decision Making for Emergency Vehicles Based on A Combinational Learning Method","date":"2020-09-09","arxiv_id":"2009.04203","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-hybrid-pac-reinforcement-learning-algorithm","title":"A Hybrid PAC Reinforcement Learning Algorithm","date":"2020-09-05","arxiv_id":"2009.02602","n_code_links":0,"syntology":null},{"paper":null,"slug":"pac-reinforcement-learning-algorithm-for","title":"PAC Reinforcement Learning Algorithm for General-Sum Markov Games","date":"2020-09-05","arxiv_id":"2009.02605","n_code_links":0,"syntology":null},{"paper":"/paper/drle-decentralized-reinforcement-learning-at","slug":"drle-decentralized-reinforcement-learning-at","title":"DRLE: Decentralized Reinforcement Learning at the Edge for Traffic Light Control in the IoV","date":"2020-09-03","arxiv_id":"2009.01502","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-nash-equilibria-in-zero-sum","title":"Learning Nash Equilibria in Zero-Sum Stochastic Games via Entropy-Regularized Policy Approximation","date":"2020-09-01","arxiv_id":"2009.00162","n_code_links":0,"syntology":null},{"paper":null,"slug":"solving-the-single-track-train-scheduling","title":"Solving the single-track train scheduling problem via Deep Reinforcement Learning","date":"2020-09-01","arxiv_id":"2009.00433","n_code_links":0,"syntology":null},{"paper":null,"slug":"theory-of-deep-q-learning-a-dynamical-systems","title":"Deep Q-Learning: Theoretical Insights from an Asymptotic Analysis","date":"2020-08-25","arxiv_id":"2008.10870","n_code_links":0,"syntology":null},{"paper":"/paper/table2charts-learning-shared-representations","slug":"table2charts-learning-shared-representations","title":"Table2Charts: Recommending Charts by Learning Shared Table Representations","date":"2020-08-24","arxiv_id":"2008.11015","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-adaptive-synchronization-approach-for","title":"An adaptive synchronization approach for weights of deep reinforcement learning","date":"2020-08-16","arxiv_id":"2008.06973","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-reinforcement-learning-based-multi-agent","title":"The reinforcement learning-based multi-agent cooperative approach for the adaptive speed regulation on a metallurgical pickling line","date":"2020-08-16","arxiv_id":"2008.06933","n_code_links":0,"syntology":null},{"paper":null,"slug":"chrome-dino-run-using-reinforcement-learning","title":"Chrome Dino Run using Reinforcement Learning","date":"2020-08-15","arxiv_id":"2008.06799","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-with-quantum","slug":"reinforcement-learning-with-quantum","title":"Reinforcement Learning with Quantum Variational Circuits","date":"2020-08-15","arxiv_id":"2008.07524","n_code_links":2,"syntology":null},{"paper":null,"slug":"decision-making-at-unsignalized-intersection","title":"Decision-making at Unsignalized Intersection for Autonomous Vehicles: Left-turn Maneuver with Deep Reinforcement Learning","date":"2020-08-14","arxiv_id":"2008.06595","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalized-radio-environment-monitoring-for","title":"Generalized Radio Environment Monitoring for Next Generation Wireless Networks","date":"2020-08-14","arxiv_id":"2008.06203","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-agent-double-deep-q-learning-for","title":"Multi-Agent Double Deep Q-Learning for Beamforming in mmWave MIMO Networks","date":"2020-08-13","arxiv_id":"2008.05943","n_code_links":0,"syntology":null},{"paper":null,"slug":"caching-placement-and-resource-allocation-for","title":"Caching Placement and Resource Allocation for Cache-Enabling UAV NOMA Networks","date":"2020-08-12","arxiv_id":"2008.05168","n_code_links":0,"syntology":null},{"paper":null,"slug":"convex-q-learning-part-1-deterministic","title":"Convex Q-Learning, Part 1: Deterministic Optimal Control","date":"2020-08-08","arxiv_id":"2008.03559","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-load-models-and-their-impacts-on","title":"Evaluating Load Models and Their Impacts on Power Transfer Limits","date":"2020-08-07","arxiv_id":"2008.03336","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-q-network-based-multi-agent","title":"Deep Q-Network Based Multi-agent Reinforcement Learning with Binary Action Agents","date":"2020-08-06","arxiv_id":"2008.04109","n_code_links":0,"syntology":null},{"paper":"/paper/deep-inverse-q-learning-with-constraints","slug":"deep-inverse-q-learning-with-constraints","title":"Deep Inverse Q-learning with Constraints","date":"2020-08-04","arxiv_id":"2008.01712","n_code_links":2,"syntology":null},{"paper":null,"slug":"cooperative-control-of-mobile-robots-with","title":"Cooperative Control of Mobile Robots with Stackelberg Learning","date":"2020-08-03","arxiv_id":"2008.00679","n_code_links":0,"syntology":null},{"paper":"/paper/qplex-duplex-dueling-multi-agent-q-learning","slug":"qplex-duplex-dueling-multi-agent-q-learning","title":"QPLEX: Duplex Dueling Multi-Agent Q-Learning","date":"2020-08-03","arxiv_id":"2008.01062","n_code_links":6,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wjh720/QPLEX"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"momentum-q-learning-with-finite-sample","title":"Momentum Q-learning with Finite-Sample Convergence Guarantee","date":"2020-07-30","arxiv_id":"2007.15418","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-dynamic-2","title":"Deep Reinforcement Learning for Dynamic Spectrum Sensing and Aggregation in Multi-Channel Wireless Networks","date":"2020-07-28","arxiv_id":"2007.13965","n_code_links":0,"syntology":null},{"paper":null,"slug":"variance-reduction-for-deep-q-learning-using","title":"Variance Reduction for Deep Q-Learning using Stochastic Recursive Gradient","date":"2020-07-25","arxiv_id":"2007.12817","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-ai-based-intrusion","title":"A Comparative Study of AI-based Intrusion Detection Techniques in Critical Infrastructures","date":"2020-07-24","arxiv_id":"2008.00088","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-vs-deep-bayesian-reinforcement-learning","title":"Trade-off on Sim2Real Learning: Real-world Learning Faster than Simulations","date":"2020-07-21","arxiv_id":"2007.10675","n_code_links":0,"syntology":null},{"paper":null,"slug":"emaq-expected-max-q-learning-operator-for","title":"EMaQ: Expected-Max Q-Learning Operator for Simple Yet Effective Offline and Online RL","date":"2020-07-21","arxiv_id":"2007.11091","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-machine-learning-approach-for-task-and","title":"A Machine Learning Approach for Task and Resource Allocation in Mobile Edge Computing Based Networks","date":"2020-07-20","arxiv_id":"2007.10102","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-agent-reinforcement-learning-in-1","title":"Multi-agent Reinforcement Learning in Bayesian Stackelberg Markov Games for Adaptive Moving Target Defense","date":"2020-07-20","arxiv_id":"2007.10457","n_code_links":0,"syntology":null},{"paper":null,"slug":"drift-deep-reinforcement-learning-for","title":"DRIFT: Deep Reinforcement Learning for Functional Software Testing","date":"2020-07-16","arxiv_id":"2007.08220","n_code_links":0,"syntology":null},{"paper":null,"slug":"meta-gradient-reinforcement-learning-with-an","title":"Meta-Gradient Reinforcement Learning with an Objective Discovered Online","date":"2020-07-16","arxiv_id":"2007.08433","n_code_links":0,"syntology":null},{"paper":null,"slug":"mixture-of-step-returns-in-bootstrapped-dqn","title":"Mixture of Step Returns in Bootstrapped DQN","date":"2020-07-16","arxiv_id":"2007.08229","n_code_links":0,"syntology":null},{"paper":"/paper/pc-pg-policy-cover-directed-exploration-for","slug":"pc-pg-policy-cover-directed-exploration-for","title":"PC-PG: Policy Cover Directed Exploration for Provable Policy Gradient Learning","date":"2020-07-16","arxiv_id":"2007.08459","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":5,"n_instrument":1,"unverified":3,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":null,"slug":"reinforcement-learning-enabled-decision","title":"Reinforcement Learning-Enabled Decision-Making Strategies for a Vehicle-Cyber-Physical-System in Connected Environment","date":"2020-07-16","arxiv_id":"2007.09101","n_code_links":0,"syntology":null},{"paper":null,"slug":"analysis-of-q-learning-with-adaptation-and","title":"Analysis of Q-learning with Adaptation and Momentum Restart for Gradient Descent","date":"2020-07-15","arxiv_id":"2007.07422","n_code_links":0,"syntology":null},{"paper":null,"slug":"qgraph-bounded-q-learning-stabilizing-model-1","title":"Qgraph-bounded Q-learning: Stabilizing Model-Free Off-Policy Deep Reinforcement Learning","date":"2020-07-15","arxiv_id":"2007.07582","n_code_links":0,"syntology":null},{"paper":"/paper/single-partition-adaptive-q-learning","slug":"single-partition-adaptive-q-learning","title":"Single-partition adaptive Q-learning","date":"2020-07-14","arxiv_id":"2007.06741","n_code_links":1,"syntology":null},{"paper":"/paper/revisiting-fundamentals-of-experience-replay","slug":"revisiting-fundamentals-of-experience-replay","title":"Revisiting Fundamentals of Experience Replay","date":"2020-07-13","arxiv_id":"2007.06700","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"simulating-multi-exit-evacuation-using-deep","title":"Simulating multi-exit evacuation using deep reinforcement learning","date":"2020-07-11","arxiv_id":"2007.05783","n_code_links":0,"syntology":null},{"paper":"/paper/provably-efficient-double-q-learning","slug":"provably-efficient-double-q-learning","title":"The Mean-Squared Error of Double Q-Learning","date":"2020-07-09","arxiv_id":"2007.05034","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wentaoweng/The-Mean-Squared-Error-of-Double-Q-Learning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/sunrise-a-simple-unified-framework-for","slug":"sunrise-a-simple-unified-framework-for","title":"SUNRISE: A Simple Unified Framework for Ensemble Learning in Deep Reinforcement Learning","date":"2020-07-09","arxiv_id":"2007.04938","n_code_links":1,"syntology":null},{"paper":null,"slug":"auto-map-a-dqn-framework-for-exploring","title":"Auto-MAP: A DQN Framework for Exploring Distributed Execution Plans for DNN Workloads","date":"2020-07-08","arxiv_id":"2007.04069","n_code_links":0,"syntology":null},{"paper":null,"slug":"cognitive-radio-network-throughput","title":"Cognitive Radio Network Throughput Maximization with Deep Reinforcement Learning","date":"2020-07-07","arxiv_id":"2007.03165","n_code_links":0,"syntology":null},{"paper":"/paper/neural-interactive-collaborative-filtering","slug":"neural-interactive-collaborative-filtering","title":"Neural Interactive Collaborative Filtering","date":"2020-07-04","arxiv_id":"2007.02095","n_code_links":1,"syntology":null}],"record_sha256":"156f1c4d26c50f9964422e6f2caf0dfc4e943718b48d3f8abd779cbacd167b54","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}