{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/q-learning/papers/15","list_of":"/task/q-learning","task":"Q-Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":15,"pages_in_order":20,"rows_per_page":100,"rows":[1401,1500],"of":1918,"counts":{"archive_papers_tagged":1918,"with_a_code_link":463,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1918,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":102,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":102,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/q-learning","prev":"/task/q-learning/papers/14","next":"/task/q-learning/papers/16","papers":[{"url":null,"slug":"logistic-q-learning","title":"Logistic Q-Learning","date":"2020-10-21","arxiv_id":"2010.11151","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-information-asymmetry-in-competitive-multi","title":"On Information Asymmetry in Competitive Multi-Agent Reinforcement Learning: Convergence and Optimality","date":"2020-10-21","arxiv_id":"2010.10901","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-using-deep-q-networks","title":"Reinforcement learning using Deep Q Networks and Q learning accurately localizes brain tumors on MRI with very small training sets","date":"2020-10-21","arxiv_id":"2010.10763","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-inference-with-multi-head-automata","title":"Language Inference with Multi-head Automata through Reinforcement Learning","date":"2020-10-20","arxiv_id":"2010.10141","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dexterous-manipulation-from","title":"Learning Dexterous Manipulation from Suboptimal Experts","date":"2020-10-16","arxiv_id":"2010.08587","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-nesterov-s-accelerated-quasi-newton-method","title":"A Nesterov's Accelerated quasi-Newton method for Global Routing using Deep Reinforcement Learning","date":"2020-10-15","arxiv_id":"2010.09465","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-type","title":"Model-Based Reinforcement Learning for Type 1Diabetes Blood Glucose Control","date":"2020-10-13","arxiv_id":"2010.06266","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameterized-reinforcement-learning-for","title":"Parameterized Reinforcement Learning for Optical System Optimization","date":"2020-10-09","arxiv_id":"2010.05769","repositories_listed":0,"syntology":null},{"url":null,"slug":"fictitious-play-in-zero-sum-stochastic-games","title":"Fictitious play in zero-sum stochastic games","date":"2020-10-08","arxiv_id":"2010.04223","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-regret-bounds-for-model-free-rl-1","title":"Model-Free Non-Stationary RL: Near-Optimal Regret and Applications in Multi-Agent RL and Inventory Control","date":"2020-10-07","arxiv_id":"2010.03161","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-empowered-trajectory-and","title":"Machine Learning Empowered Trajectory and Passive Beamforming Design in UAV-RIS Wireless Networks","date":"2020-10-06","arxiv_id":"2010.02749","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-learning-in-deep-q-networks","title":"Cross Learning in Deep Q-Networks","date":"2020-09-29","arxiv_id":"2009.13780","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-analysis-for-double-q-learning","title":"Finite-Time Analysis for Double Q-learning","date":"2020-09-29","arxiv_id":"2009.14257","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-jump-q-evaluation-for-offline-policy","title":"Deep Jump Q-Evaluation for Offline Policy Evaluation in Continuous Action Space","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-regret-bounds-for-model-free-rl","title":"Near-Optimal Regret Bounds for Model-Free RL in Non-Stationary Episodic MDPs","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-linear-value-1","title":"Towards Understanding Linear Value Decomposition in Cooperative Multi-Agent Q-Learning","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-approach-for-tactical-decision-making","title":"A New Approach for Tactical Decision Making in Lane Changing: Sample Efficient Deep Q Learning with a Safety Feedback Reward","date":"2020-09-24","arxiv_id":"2009.11905","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-q-learning-provably-efficient-an-extended","title":"Is Q-Learning Provably Efficient? An Extended Analysis","date":"2020-09-22","arxiv_id":"2009.10396","repositories_listed":0,"syntology":null},{"url":null,"slug":"hidden-incentives-for-auto-induced","title":"Hidden Incentives for Auto-Induced Distributional Shift","date":"2020-09-19","arxiv_id":"2009.09153","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-dynamic-resource","title":"Reinforcement Learning for Dynamic Resource Optimization in 5G Radio Access Network Slicing","date":"2020-09-14","arxiv_id":"2009.06579","repositories_listed":0,"syntology":null},{"url":null,"slug":"aoi-minimization-in-status-update-control","title":"AoI Minimization in Status Update Control with Energy Harvesting Sensors","date":"2020-09-09","arxiv_id":"2009.04224","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-option","title":"Deep Reinforcement Learning for Option Replication and Hedging","date":"2020-09-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hybrid-pac-reinforcement-learning-algorithm","title":"A Hybrid PAC Reinforcement Learning Algorithm","date":"2020-09-05","arxiv_id":"2009.02602","repositories_listed":0,"syntology":null},{"url":null,"slug":"pac-reinforcement-learning-algorithm-for","title":"PAC Reinforcement Learning Algorithm for General-Sum Markov Games","date":"2020-09-05","arxiv_id":"2009.02605","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-machine-teaching-to-investigate-human","title":"Using Machine Teaching to Investigate Human Assumptions when Teaching Reinforcement Learners","date":"2020-09-05","arxiv_id":"2009.02476","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-nash-equilibria-in-zero-sum","title":"Learning Nash Equilibria in Zero-Sum Stochastic Games via Entropy-Regularized Policy Approximation","date":"2020-09-01","arxiv_id":"2009.00162","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-the-single-track-train-scheduling","title":"Solving the single-track train scheduling problem via Deep Reinforcement Learning","date":"2020-09-01","arxiv_id":"2009.00433","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-policy-evaluation-for-value-based","title":"Inverse Policy Evaluation for Value-based Sequential Decision-making","date":"2020-08-26","arxiv_id":"2008.11329","repositories_listed":0,"syntology":null},{"url":null,"slug":"theory-of-deep-q-learning-a-dynamical-systems","title":"Deep Q-Learning: Theoretical Insights from an Asymptotic Analysis","date":"2020-08-25","arxiv_id":"2008.10870","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-reinforcement-learning-based-multi-agent","title":"The reinforcement learning-based multi-agent cooperative approach for the adaptive speed regulation on a metallurgical pickling line","date":"2020-08-16","arxiv_id":"2008.06933","repositories_listed":0,"syntology":null},{"url":null,"slug":"chrome-dino-run-using-reinforcement-learning","title":"Chrome Dino Run using Reinforcement Learning","date":"2020-08-15","arxiv_id":"2008.06799","repositories_listed":0,"syntology":null},{"url":null,"slug":"decision-making-at-unsignalized-intersection","title":"Decision-making at Unsignalized Intersection for Autonomous Vehicles: Left-turn Maneuver with Deep Reinforcement Learning","date":"2020-08-14","arxiv_id":"2008.06595","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-double-deep-q-learning-for","title":"Multi-Agent Double Deep Q-Learning for Beamforming in mmWave MIMO Networks","date":"2020-08-13","arxiv_id":"2008.05943","repositories_listed":0,"syntology":null},{"url":null,"slug":"caching-placement-and-resource-allocation-for","title":"Caching Placement and Resource Allocation for Cache-Enabling UAV NOMA Networks","date":"2020-08-12","arxiv_id":"2008.05168","repositories_listed":0,"syntology":null},{"url":null,"slug":"convex-q-learning-part-1-deterministic","title":"Convex Q-Learning, Part 1: Deterministic Optimal Control","date":"2020-08-08","arxiv_id":"2008.03559","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-load-models-and-their-impacts-on","title":"Evaluating Load Models and Their Impacts on Power Transfer Limits","date":"2020-08-07","arxiv_id":"2008.03336","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-network-based-multi-agent","title":"Deep Q-Network Based Multi-agent Reinforcement Learning with Binary Action Agents","date":"2020-08-06","arxiv_id":"2008.04109","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-analysis-of-deep-reinforcement","title":"A Comparative Analysis of Deep Reinforcement Learning-enabled Freeway Decision-making for Automated Vehicles","date":"2020-08-04","arxiv_id":"2008.01302","repositories_listed":0,"syntology":null},{"url":null,"slug":"gencos-behaviors-modeling-based-on-q-learning","title":"GenCos' Behaviors Modeling Based on Q Learning Improved by Dichotomy","date":"2020-08-04","arxiv_id":"2008.01536","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-control-of-mobile-robots-with","title":"Cooperative Control of Mobile Robots with Stackelberg Learning","date":"2020-08-03","arxiv_id":"2008.00679","repositories_listed":0,"syntology":null},{"url":null,"slug":"momentum-q-learning-with-finite-sample","title":"Momentum Q-learning with Finite-Sample Convergence Guarantee","date":"2020-07-30","arxiv_id":"2007.15418","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-dynamic-2","title":"Deep Reinforcement Learning for Dynamic Spectrum Sensing and Aggregation in Multi-Channel Wireless Networks","date":"2020-07-28","arxiv_id":"2007.13965","repositories_listed":0,"syntology":null},{"url":null,"slug":"variance-reduction-for-deep-q-learning-using","title":"Variance Reduction for Deep Q-Learning using Stochastic Recursive Gradient","date":"2020-07-25","arxiv_id":"2007.12817","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-of-ai-based-intrusion","title":"A Comparative Study of AI-based Intrusion Detection Techniques in Critical Infrastructures","date":"2020-07-24","arxiv_id":"2008.00088","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-vs-deep-bayesian-reinforcement-learning","title":"Trade-off on Sim2Real Learning: Real-world Learning Faster than Simulations","date":"2020-07-21","arxiv_id":"2007.10675","repositories_listed":0,"syntology":null},{"url":null,"slug":"emaq-expected-max-q-learning-operator-for","title":"EMaQ: Expected-Max Q-Learning Operator for Simple Yet Effective Offline and Online RL","date":"2020-07-21","arxiv_id":"2007.11091","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-machine-learning-approach-for-task-and","title":"A Machine Learning Approach for Task and Resource Allocation in Mobile Edge Computing Based Networks","date":"2020-07-20","arxiv_id":"2007.10102","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-in-1","title":"Multi-agent Reinforcement Learning in Bayesian Stackelberg Markov Games for Adaptive Moving Target Defense","date":"2020-07-20","arxiv_id":"2007.10457","repositories_listed":0,"syntology":null},{"url":null,"slug":"same-day-delivery-with-fairness","title":"Same-Day Delivery with Fairness","date":"2020-07-19","arxiv_id":"2007.09541","repositories_listed":0,"syntology":null},{"url":null,"slug":"drift-deep-reinforcement-learning-for","title":"DRIFT: Deep Reinforcement Learning for Functional Software Testing","date":"2020-07-16","arxiv_id":"2007.08220","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-gradient-reinforcement-learning-with-an","title":"Meta-Gradient Reinforcement Learning with an Objective Discovered Online","date":"2020-07-16","arxiv_id":"2007.08433","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-enabled-decision","title":"Reinforcement Learning-Enabled Decision-Making Strategies for a Vehicle-Cyber-Physical-System in Connected Environment","date":"2020-07-16","arxiv_id":"2007.09101","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-q-learning-with-adaptation-and","title":"Analysis of Q-learning with Adaptation and Momentum Restart for Gradient Descent","date":"2020-07-15","arxiv_id":"2007.07422","repositories_listed":0,"syntology":null},{"url":null,"slug":"qgraph-bounded-q-learning-stabilizing-model-1","title":"Qgraph-bounded Q-learning: Stabilizing Model-Free Off-Policy Deep Reinforcement Learning","date":"2020-07-15","arxiv_id":"2007.07582","repositories_listed":0,"syntology":null},{"url":null,"slug":"hedging-using-reinforcement-learning","title":"Hedging using reinforcement learning: Contextual $k$-Armed Bandit versus $Q$-learning","date":"2020-07-03","arxiv_id":"2007.01623","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularly-updated-deterministic-policy","title":"Regularly Updated Deterministic Policy Gradient Algorithm","date":"2020-07-01","arxiv_id":"2007.00169","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-more-efficient-q-learning-in-the","title":"Provably More Efficient Q-Learning in the One-Sided-Feedback/Full-Feedback Settings","date":"2020-06-30","arxiv_id":"2007.00080","repositories_listed":0,"syntology":null},{"url":null,"slug":"concept-and-the-implementation-of-a-tool-to","title":"Concept and the implementation of a tool to convert industry 4.0 environments modeled as FSM to an OpenAI Gym wrapper","date":"2020-06-29","arxiv_id":"2006.16035","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-reinforcement-learning-to-herd-a","title":"Using Reinforcement Learning to Herd a Robotic Swarm to a Target Distribution","date":"2020-06-29","arxiv_id":"2006.15807","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-finite-reward-automaton-inference-and","title":"Active Finite Reward Automaton Inference and Reinforcement Learning Using Queries and Counterexamples","date":"2020-06-28","arxiv_id":"2006.15714","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-handwritten","title":"Reinforcement Learning Based Handwritten Digit Recognition with Two-State Q-Learning","date":"2020-06-28","arxiv_id":"2007.01193","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-with-differential-entropy-of-q","title":"Q-Learning with Differential Entropy of Q-Tables","date":"2020-06-26","arxiv_id":"2006.14795","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-network-driven-catheter-segmentation","title":"Deep Q-Network-Driven Catheter Segmentation in 3D US by Hybrid Constrained Semi-Supervised Learning and Dual-UNet","date":"2020-06-25","arxiv_id":"2006.14702","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-minimization-in-uav-aided-networks","title":"Energy Minimization in UAV-Aided Networks: Actor-Critic Learning for Constrained Scheduling Optimization","date":"2020-06-24","arxiv_id":"2006.13610","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-overestimation-bias-by-increasing","title":"Preventing Value Function Collapse in Ensemble {Q}-Learning by Maximizing Representation Diversity","date":"2020-06-24","arxiv_id":"2006.13823","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-reinforcement-q-learning-for-mean","title":"Unified Reinforcement Q-Learning for Mean Field Game and Control Problems","date":"2020-06-24","arxiv_id":"2006.13912","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-control-for-radar","title":"Deep Reinforcement Learning Control for Radar Detection and Tracking in Congested Spectral Environments","date":"2020-06-23","arxiv_id":"2006.13173","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-reinforcement-learning-with-self","title":"Near-Optimal Reinforcement Learning with Self-Play","date":"2020-06-22","arxiv_id":"2006.12007","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-reinforcement-learning-near","title":"Risk-Sensitive Reinforcement Learning: Near-Optimal Risk-Sample Tradeoff in Regret","date":"2020-06-22","arxiv_id":"2006.13827","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybridizing-the-1-5-th-success-rule-with-q","title":"Hybridizing the 1/5-th Success Rule with Q-Learning for Controlling the Mutation Rate of an Evolutionary Algorithm","date":"2020-06-19","arxiv_id":"2006.11026","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameterized-mdps-and-reinforcement-learning","title":"Parameterized MDPs and Reinforcement Learning Problems -- A Maximum Entropy Principle Based Framework","date":"2020-06-17","arxiv_id":"2006.09646","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-with-logarithmic-regret","title":"$Q$-learning with Logarithmic Regret","date":"2020-06-16","arxiv_id":"2006.09118","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-teaching-dimension-of-q-learning","title":"The Sample Complexity of Teaching-by-Reinforcement on Q-Learning","date":"2020-06-16","arxiv_id":"2006.09324","repositories_listed":0,"syntology":null},{"url":null,"slug":"runtime-adaptation-in-wireless-sensor-nodes","title":"Runtime Adaptation in Wireless Sensor Nodes Using Structured Learning","date":"2020-06-15","arxiv_id":"2006.08666","repositories_listed":0,"syntology":null},{"url":null,"slug":"decorrelated-double-q-learning","title":"Decorrelated Double Q-learning","date":"2020-06-12","arxiv_id":"2006.06956","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-neural","title":"Deep Reinforcement Learning for Neural Control","date":"2020-06-12","arxiv_id":"2006.07352","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-and-multi-agent-collaboration-in-a","title":"Human and Multi-Agent collaboration in a human-MARL teaming framework","date":"2020-06-12","arxiv_id":"2006.07301","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-guaranteed-reinforcement-learning","title":"Safety-guaranteed Reinforcement Learning based on Multi-class Support Vector Machine","date":"2020-06-12","arxiv_id":"2006.07446","repositories_listed":0,"syntology":null},{"url":"/paper/self-imitation-learning-via-generalized-lower","slug":"self-imitation-learning-via-generalized-lower","title":"Self-Imitation Learning via Generalized Lower Bound Q-learning","date":"2020-06-12","arxiv_id":"2006.07442","repositories_listed":0,"syntology":{"n":20,"n_ran":16,"n_constructed":4,"n_ran_checked":13,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":2,"n_no_contract":11,"n_pointer_only":11,"phrase":"16 ran (of which 4 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 2 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/self-imitation-learning-via-generalized-lower#ran","syntology_url":"https://syntology.ai/paper/2006.07442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.07442"}},"official":null}},{"url":null,"slug":"exploration-by-maximizing-renyi-entropy-for","title":"Exploration by Maximizing Rényi Entropy for Reward-Free RL Framework","date":"2020-06-11","arxiv_id":"2006.06193","repositories_listed":0,"syntology":null},{"url":null,"slug":"zeroth-order-supervised-policy-improvement","title":"Zeroth-Order Supervised Policy Improvement","date":"2020-06-11","arxiv_id":"2006.06600","repositories_listed":0,"syntology":null},{"url":null,"slug":"fitted-q-learning-for-relational-domains","title":"Fitted Q-Learning for Relational Domains","date":"2020-06-10","arxiv_id":"2006.05595","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-algorithm-and-regret-analysis-for-1","title":"Model-Free Algorithm and Regret Analysis for MDPs with Long-Term Constraints","date":"2020-06-10","arxiv_id":"2006.05961","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-in-a","title":"Multi-Agent Reinforcement Learning in a Realistic Limit Order Book Market Simulation","date":"2020-06-10","arxiv_id":"2006.05574","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-cost-management-in-smart-meters-an","title":"Privacy-Cost Management in Smart Meters with Mutual Information-Based Reinforcement Learning","date":"2020-06-10","arxiv_id":"2006.06106","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-greedyucb-a-new-exploration-policy-for","title":"Q-greedyUCB: a New Exploration Policy for Adaptive and Resource-efficient Scheduling","date":"2020-06-10","arxiv_id":"2006.05902","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-reinforcement-learning","title":"Self-Supervised Reinforcement Learning for Recommender Systems","date":"2020-06-10","arxiv_id":"2006.05779","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-joint-self","title":"Reinforcement Learning-Based Joint Self-Optimisation Method for the Fuzzy Logic Handover Algorithm in 5G HetNets","date":"2020-06-09","arxiv_id":"2006.05010","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-model-free-learning-algorithm-for-infinite","title":"A Model-free Learning Algorithm for Infinite-horizon Average-reward MDPs with Near-optimal Regret","date":"2020-06-08","arxiv_id":"2006.04354","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-a-cartpole-system-with","title":"Balancing a CartPole System with Reinforcement Learning -- A Tutorial","date":"2020-06-08","arxiv_id":"2006.04938","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-temporal-difference-and-q-learning-learn","title":"Can Temporal-Difference and Q-Learning Learn Representation? A Mean-Field Theory","date":"2020-06-08","arxiv_id":"2006.04761","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-step-and-resilient-predictive-q","title":"A Multi-step and Resilient Predictive Q-learning Algorithm for IoT with Human Operators in the Loop: A Case Study in Water Supply Networks","date":"2020-06-06","arxiv_id":"2006.03899","repositories_listed":0,"syntology":null},{"url":null,"slug":"logical-team-q-learning-an-approach-towards","title":"Logical Team Q-learning: An approach towards factored policies in cooperative MARL","date":"2020-06-05","arxiv_id":"2006.03553","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-complexity-of-asynchronous-q-learning","title":"Sample Complexity of Asynchronous Q-Learning: Sharper Analysis and Variance Reduction","date":"2020-06-04","arxiv_id":"2006.03041","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-bias-in-face-recognition-using","title":"Mitigating Bias in Face Recognition Using Skewness-Aware Reinforcement Learning","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-architecture-search-with-reinforce-and","title":"Hyperparameter optimization with REINFORCE and Transformers","date":"2020-06-01","arxiv_id":"2006.00939","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-linear-value","title":"Towards Understanding Cooperative Multi-Agent Q-Learning with Value Factorization","date":"2020-05-31","arxiv_id":"2006.00587","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-joint-user-ap-association-and-1","title":"Learning-Based Joint User-AP Association and Resource Allocation in Ultra Dense Network","date":"2020-05-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"active-measure-reinforcement-learning-for","title":"Active Measure Reinforcement Learning for Observation Cost Minimization","date":"2020-05-26","arxiv_id":"2005.12697","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-power-1","title":"Deep Reinforcement Learning Based Power Allocation for D2D Network","date":"2020-05-25","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"e20737993dbfd47dbe1b344f5afc534baa9b642ce627e873c8a4da34062aca4b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}