{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/q-learning/papers/10","list_of":"/task/q-learning","task":"Q-Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":20,"rows_per_page":100,"rows":[901,1000],"of":1918,"counts":{"archive_papers_tagged":1918,"with_a_code_link":463,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1918,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":102,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":102,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/q-learning","prev":"/task/q-learning/papers/9","next":"/task/q-learning/papers/11","papers":[{"url":null,"slug":"pruning-the-way-to-reliable-policies-a-multi","title":"Pruning the Way to Reliable Policies: A Multi-Objective Deep Q-Learning Approach to Critical Care","date":"2023-06-13","arxiv_id":"2306.08044","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-information-state-based","title":"Approximate information state based convergence analysis of recurrent Q-learning","date":"2023-06-09","arxiv_id":"2306.05991","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-analysis-of-minimax-q-learning","title":"Finite-Time Analysis of Minimax Q-Learning for Two-Player Zero-Sum Markov Games: Switching System Approach","date":"2023-06-09","arxiv_id":"2306.05700","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-inference-in-hebbian-learning-networks","title":"Active Inference in Hebbian Learning Networks","date":"2023-06-08","arxiv_id":"2306.05053","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-control-of-4","title":"Reinforcement Learning-Based Control of CrazyFlie 2.X Quadrotor","date":"2023-06-06","arxiv_id":"2306.03951","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-learning-versus-proximal-policy","title":"Deep Q-Learning versus Proximal Policy Optimization: Performance Comparison in a Material Sorting Task","date":"2023-06-02","arxiv_id":"2306.01451","repositories_listed":0,"syntology":null},{"url":null,"slug":"iql-td-mpc-implicit-q-learning-for","title":"IQL-TD-MPC: Implicit Q-Learning for Hierarchical Model Predictive Control","date":"2023-06-01","arxiv_id":"2306.00867","repositories_listed":0,"syntology":null},{"url":null,"slug":"va-learning-as-a-more-efficient-alternative","title":"VA-learning as a more efficient alternative to Q-learning","date":"2023-05-29","arxiv_id":"2305.18161","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-complexity-of-variance-reduced","title":"Sample Complexity of Variance-reduced Distributionally Robust Q-learning","date":"2023-05-28","arxiv_id":"2305.18420","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-analysis-of-portfolio","title":"A Comparative Analysis of Portfolio Optimization Using Mean-Variance, Hierarchical Risk Parity, and Reinforcement Learning Approaches on the Indian Stock Market","date":"2023-05-27","arxiv_id":"2305.17523","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-reward-machines","title":"Reinforcement Learning With Reward Machines in Stochastic Games","date":"2023-05-27","arxiv_id":"2305.17372","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-in-3","title":"Sample Efficient Reinforcement Learning in Mixed Systems through Augmented Samples and Its Applications to Queueing Networks","date":"2023-05-25","arxiv_id":"2305.16483","repositories_listed":0,"syntology":null},{"url":null,"slug":"2305-14656","title":"RSRM: Reinforcement Symbolic Regression Machine","date":"2023-05-24","arxiv_id":"2305.14656","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-experience-replay-for-continual","title":"OER: Offline Experience Replay for Continual Offline Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.13804","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-provably-stable-and","title":"A Framework for Provably Stable and Consistent Training of Deep Feedforward Networks","date":"2023-05-20","arxiv_id":"2305.12125","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-risk-averse-q-learning-with","title":"Bayesian Risk-Averse Q-Learning with Streaming Observations","date":"2023-05-18","arxiv_id":"2305.11300","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-blessing-of-heterogeneity-in-federated-q","title":"The Blessing of Heterogeneity in Federated Q-Learning: Linear Speedup and Beyond","date":"2023-05-18","arxiv_id":"2305.10697","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-robust-average-reward","title":"Model-Free Robust Average-Reward Reinforcement Learning","date":"2023-05-17","arxiv_id":"2305.10504","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-home-energy-management-vae-gan","title":"Smart Home Energy Management: VAE-GAN synthetic dataset generator and Q-learning","date":"2023-05-14","arxiv_id":"2305.08885","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-practical-robust-reinforcement-learning","title":"On Practical Robust Reinforcement Learning: Practical Uncertainty Set and Double-Agent Algorithm","date":"2023-05-11","arxiv_id":"2305.06657","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-learning-based-distribution-network","title":"Deep Q-Learning-based Distribution Network Reconfiguration for Reliability Improvement","date":"2023-05-02","arxiv_id":"2305.01180","repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-quantum-reinforcement-learning","title":"BCQQ: Batch-Constraint Quantum Q-Learning with Cyclic Data Re-uploading","date":"2023-04-27","arxiv_id":"2305.00905","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-q-learning-for-continuous-time-linear","title":"Safe Q-learning for continuous-time linear systems","date":"2023-04-26","arxiv_id":"2304.13573","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-services-function-chain","title":"Adaptive Services Function Chain Orchestration For Digital Health Twin Use Cases: Heuristic-boosted Q-Learning Approach","date":"2023-04-25","arxiv_id":"2304.12853","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-based-equilibria","title":"Learned Collusion","date":"2023-04-25","arxiv_id":"2304.12647","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-exploration-for-effective-multi-agent-q","title":"Graph Exploration for Effective Multi-agent Q-Learning","date":"2023-04-19","arxiv_id":"2304.09547","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-deep-q-learning-with-distributed","title":"Quantum deep Q learning with distributed prioritized experience replay","date":"2023-04-19","arxiv_id":"2304.09648","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-a-q-learning-algorithm-application","title":"A study on a Q-Learning algorithm application to a manufacturing assembly problem","date":"2023-04-17","arxiv_id":"2304.08375","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-decision-making-in-spatial-learning-a","title":"Exploring the Noise Resilience of Successor Features and Predecessor Features Algorithms in One and Two-Dimensional Environments","date":"2023-04-14","arxiv_id":"2304.06894","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-applied-to-an","title":"Deep reinforcement learning applied to an assembly sequence planning problem with user preferences","date":"2023-04-13","arxiv_id":"2304.06567","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-minimum-state","title":"Reinforcement Learning Based Minimum State-flipped Control for the Reachability of Boolean Control Networks","date":"2023-04-11","arxiv_id":"2304.04950","repositories_listed":0,"syntology":null},{"url":null,"slug":"rels-dqn-a-robust-and-efficient-local-search","title":"RELS-DQN: A Robust and Efficient Local Search Framework for Combinatorial Optimization","date":"2023-04-11","arxiv_id":"2304.06048","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-optimal-1","title":"Deep Reinforcement Learning Based Optimal Infinite-Horizon Control of Probabilistic Boolean Control Networks","date":"2023-04-07","arxiv_id":"2304.03489","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-gradient-deep-reinforcement-learning-for","title":"Full Gradient Deep Reinforcement Learning for Average-Reward Criterion","date":"2023-04-07","arxiv_id":"2304.03729","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-tutorial-introduction-to-reinforcement","title":"A Tutorial Introduction to Reinforcement Learning","date":"2023-04-03","arxiv_id":"2304.00803","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantitative-trading-using-deep-q-learning","title":"Quantitative Trading using Deep Q Learning","date":"2023-04-03","arxiv_id":"2304.06037","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-reinforcement-learning","title":"Understanding Reinforcement Learning Algorithms: The Progress from Basic Q-learning to Proximal Policy Optimization","date":"2023-03-31","arxiv_id":"2304.00026","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-based-system-for-path-planning","title":"Q-Learning based system for path planning with unmanned aerial vehicles swarms in obstacle environments","date":"2023-03-30","arxiv_id":"2303.17655","repositories_listed":0,"syntology":null},{"url":null,"slug":"concentration-of-contractive-stochastic-1","title":"Concentration of Contractive Stochastic Approximation: Additive and Multiplicative Noise","date":"2023-03-28","arxiv_id":"2303.15740","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-multi-agent-deep-q-learning-for","title":"Distributed Multi-Agent Deep Q-Learning for Fast Roaming in IEEE 802.11ax Wi-Fi Systems","date":"2023-03-25","arxiv_id":"2304.01210","repositories_listed":0,"syntology":null},{"url":null,"slug":"specific-investments-under-negotiated","title":"Specific investments under negotiated transfer pricing: effects of different surplus sharing parameters on managerial performance: An agent-based simulation with fuzzy Q-learning agents","date":"2023-03-25","arxiv_id":"2303.14515","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-path-following-on-rivers-using","title":"Robust Path Following on Rivers Using Bootstrapped Reinforcement Learning","date":"2023-03-24","arxiv_id":"2303.15178","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-intelligence-and-dual-contract","title":"Artificial Intelligence and Dual Contract","date":"2023-03-22","arxiv_id":"2303.12350","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-nars-and-reinforcement-learning-an","title":"Comparing NARS and Reinforcement Learning: An Analysis of ONA and $Q$-Learning Algorithms","date":"2023-03-17","arxiv_id":"2304.03291","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-safe-propofol-dosing-during-general","title":"Towards Real-World Applications of Personalized Anesthesia Using Policy Constraint Q Learning for Propofol Infusion Control","date":"2023-03-17","arxiv_id":"2303.10180","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-inspection-method-of-unmanned-aerial","title":"Self-Inspection Method of Unmanned Aerial Vehicles in Power Plants Using Deep Q-Network Reinforcement Learning","date":"2023-03-16","arxiv_id":"2303.09013","repositories_listed":0,"syntology":null},{"url":null,"slug":"smoothed-q-learning","title":"Smoothed Q-learning","date":"2023-03-15","arxiv_id":"2303.08631","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-tree-reconstruction-game-phylogenetic","title":"The tree reconstruction game: phylogenetic reconstruction using reinforcement learning","date":"2023-03-12","arxiv_id":"2303.06695","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-twin-assisted-knowledge-distillation","title":"Digital Twin-Assisted Knowledge Distillation Framework for Heterogeneous Federated Learning","date":"2023-03-10","arxiv_id":"2303.06155","repositories_listed":0,"syntology":null},{"url":null,"slug":"ignorance-is-bliss-robust-control-via-1","title":"Ignorance is Bliss: Robust Control via Information Gating","date":"2023-03-10","arxiv_id":"2303.06121","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-strategic-value-and-cooperation-in","title":"Learning Strategic Value and Cooperation in Multi-Player Stochastic Games through Side Payments","date":"2023-03-09","arxiv_id":"2303.05307","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-environment-transformer-and-offline","title":"Environment Transformer and Policy Optimization for Model-Based Offline Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03811","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-via-epistemic-value-estimation","title":"Exploration via Epistemic Value Estimation","date":"2023-03-07","arxiv_id":"2303.04012","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-a3c-deep-reinforcement-learning-on","title":"Double A3C: Deep Reinforcement Learning on OpenAI Gym Games","date":"2023-03-04","arxiv_id":"2303.02271","repositories_listed":0,"syntology":null},{"url":null,"slug":"wasserstein-actor-critic-directed-exploration","title":"Wasserstein Actor-Critic: Directed Exploration via Optimism for Continuous-Actions Control","date":"2023-03-04","arxiv_id":"2303.02378","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-o-ran-traffic-steering-for-urllc","title":"Intelligent O-RAN Traffic Steering for URLLC Through Deep Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.01960","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-trader-without","title":"A Deep Reinforcement Learning Trader without Offline Training","date":"2023-03-01","arxiv_id":"2303.00356","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-sample-guarantees-for-nash-q-learning","title":"Finite-sample Guarantees for Nash Q-learning with Linear Function Approximation","date":"2023-03-01","arxiv_id":"2303.00177","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-point-to-which-soft-actor-critic","title":"The Point to Which Soft Actor-Critic Converges","date":"2023-03-01","arxiv_id":"2303.01240","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimizing-the-outage-probability-in-a-markov","title":"Minimizing the Outage Probability in a Markov Decision Process","date":"2023-02-28","arxiv_id":"2302.14714","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-finite-sample-complexity-bound-for","title":"A Finite Sample Complexity Bound for Distributionally Robust Q-learning","date":"2023-02-26","arxiv_id":"2302.13203","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-cogni-an-integrated-causal-reinforcement","title":"Q-Cogni: An Integrated Causal Reinforcement Learning Framework","date":"2023-02-26","arxiv_id":"2302.13240","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-bellman-s-principle-of-optimality-and","title":"On Bellman's principle of optimality and Reinforcement learning for safety-constrained Markov decision process","date":"2023-02-25","arxiv_id":"2302.13152","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-gauss-newton-temporal","title":"Gauss-Newton Temporal Difference Learning with Nonlinear Function Approximation","date":"2023-02-25","arxiv_id":"2302.13087","repositories_listed":0,"syntology":null},{"url":null,"slug":"kernel-based-distributed-q-learning-a","title":"Kernel-Based Distributed Q-Learning: A Scalable Reinforcement Learning Approach for Dynamic Treatment Regimes","date":"2023-02-21","arxiv_id":"2302.10434","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-auto-landing-control-of-an-agile","title":"Robust Auto-landing Control of an agile Regional Jet Using Fuzzy Q-learning","date":"2023-02-21","arxiv_id":"2302.10997","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-q-learning-for-stochastic-games","title":"Logit-Q Dynamics for Efficient Learning in Stochastic Teams","date":"2023-02-20","arxiv_id":"2302.09806","repositories_listed":0,"syntology":null},{"url":null,"slug":"forecasting-and-stabilizing-chaotic-regimes","title":"Forecasting and stabilizing chaotic regimes in two macroeconomic models via artificial intelligence technologies and control methods","date":"2023-02-20","arxiv_id":"2302.12019","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-offline-reinforcement-learning-for-real","title":"Deep Offline Reinforcement Learning for Real-world Treatment Optimization Applications","date":"2023-02-15","arxiv_id":"2302.07549","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-statistical-inference-for-nonlinear","title":"Online Statistical Inference for Nonlinear Stochastic Approximation with Markovian Data","date":"2023-02-15","arxiv_id":"2302.07690","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lifetime-extended-energy-management","title":"A Lifetime Extended Energy Management Strategy for Fuel Cell Hybrid Electric Vehicles via Self-Learning Fuzzy Reinforcement Learning","date":"2023-02-13","arxiv_id":"2302.06236","repositories_listed":0,"syntology":null},{"url":null,"slug":"computation-offloading-for-uncertain-marine","title":"Computation Offloading for Uncertain Marine Tasks by Cooperation of UAVs and Vessels","date":"2023-02-13","arxiv_id":"2302.06055","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentially-private-deep-q-learning-for","title":"Differentially Private Deep Q-Learning for Pattern Privacy Preservation in MEC Offloading","date":"2023-02-09","arxiv_id":"2302.04608","repositories_listed":0,"syntology":null},{"url":null,"slug":"catch-me-if-you-can-improving-adversaries-in","title":"Catch Me If You Can: Improving Adversaries in Cyber-Security With Q-Learning Algorithms","date":"2023-02-07","arxiv_id":"2302.03768","repositories_listed":0,"syntology":null},{"url":null,"slug":"macoptions-multi-agent-learning-with","title":"MACOptions: Multi-Agent Learning with Centralized Controller and Options Framework","date":"2023-02-07","arxiv_id":"2302.03800","repositories_listed":0,"syntology":null},{"url":null,"slug":"refined-value-based-offline-rl-under","title":"Offline Minimax Soft-Q-learning Under Realizability and Partial Coverage","date":"2023-02-05","arxiv_id":"2302.02392","repositories_listed":0,"syntology":null},{"url":null,"slug":"best-possible-q-learning","title":"Best Possible Q-Learning","date":"2023-02-02","arxiv_id":"2302.01188","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversity-through-exclusion-dte-niche","title":"Diversity Through Exclusion (DTE): Niche Identification for Reinforcement Learning through Value-Decomposition","date":"2023-02-02","arxiv_id":"2302.01180","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-complexity-of-kernel-based-q-learning","title":"Sample Complexity of Kernel-Based Q-Learning","date":"2023-02-01","arxiv_id":"2302.00727","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-robustness-of-the-deep","title":"Analyzing Robustness of the Deep Reinforcement Learning Algorithm in Ramp Metering Applications Considering False Data Injection Attack and Defense","date":"2023-01-28","arxiv_id":"2301.12036","repositories_listed":0,"syntology":null},{"url":null,"slug":"rcsearcher-reaction-center-identification-in","title":"RCsearcher: Reaction Center Identification in Retrosynthesis via Deep Q-Learning","date":"2023-01-28","arxiv_id":"2301.12071","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-surplus-sharing-on-the-outcomes","title":"The impact of surplus sharing on the outcomes of specific investments under negotiated transfer pricing: An agent-based simulation with fuzzy Q-learning agents","date":"2023-01-28","arxiv_id":"2301.12255","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-trajectory-distributionally-robust","title":"Single-Trajectory Distributionally Robust Reinforcement Learning","date":"2023-01-27","arxiv_id":"2301.11721","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedhql-federated-heterogeneous-q-learning","title":"FedHQL: Federated Heterogeneous Q-Learning","date":"2023-01-26","arxiv_id":"2301.11135","repositories_listed":0,"syntology":null},{"url":null,"slug":"asymptotic-convergence-and-performance-of","title":"Asymptotic Convergence and Performance of Multi-Agent Q-Learning Dynamics","date":"2023-01-23","arxiv_id":"2301.09619","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-deep-double-duelling-q-learning","title":"Asynchronous Deep Double Duelling Q-Learning for Trading-Signal Execution in Limit Order Book Markets","date":"2023-01-20","arxiv_id":"2301.08688","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-averse-reinforcement-learning-via","title":"Risk-Averse Reinforcement Learning via Dynamic Time-Consistent Risk Measures","date":"2023-01-14","arxiv_id":"2301.05981","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-model-free-reinforcement","title":"Decentralized model-free reinforcement learning in stochastic games with average-reward objective","date":"2023-01-13","arxiv_id":"2301.05630","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-deep-q-learning-based-handover","title":"Hierarchical Deep Q-Learning Based Handover in Wireless Networks with Dual Connectivity","date":"2023-01-13","arxiv_id":"2301.05391","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-power-level-q-learning-algorithm-for","title":"Multi-Power Level $Q$-Learning Algorithm for Random Access in NOMA mMTC Systems","date":"2023-01-12","arxiv_id":"2301.05196","repositories_listed":0,"syntology":null},{"url":null,"slug":"tuning-path-tracking-controllers-for","title":"Tuning Path Tracking Controllers for Autonomous Cars Using Reinforcement Learning","date":"2023-01-09","arxiv_id":"2301.03363","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-conservative-q-learning-for","title":"Contextual Conservative Q-Learning for Offline Reinforcement Learning","date":"2023-01-03","arxiv_id":"2301.01298","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-spectral-q-learning-with-application-to","title":"Deep Spectral Q-learning with Application to Mobile Health","date":"2023-01-03","arxiv_id":"2301.00927","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-difference-learning-with-compressed","title":"Temporal Difference Learning with Compressed Updates: Error-Feedback meets Reinforcement Learning","date":"2023-01-03","arxiv_id":"2301.00944","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-surface-codes-with-deep","title":"Decoding surface codes with deep reinforcement learning and probabilistic policy reuse","date":"2022-12-22","arxiv_id":"2212.11890","repositories_listed":0,"syntology":null},{"url":null,"slug":"bandit-approach-to-conflict-free-multi-agent","title":"Bandit approach to conflict-free multi-agent Q-learning in view of photonic implementation","date":"2022-12-20","arxiv_id":"2212.09926","repositories_listed":0,"syntology":null},{"url":null,"slug":"taming-lagrangian-chaos-with-multi-objective","title":"Taming Lagrangian Chaos with Multi-Objective Reinforcement Learning","date":"2022-12-19","arxiv_id":"2212.09612","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-robot-reinforcement-learning-with","title":"Offline Robot Reinforcement Learning with Uncertainty-Guided Human Expert Sampling","date":"2022-12-16","arxiv_id":"2212.08232","repositories_listed":0,"syntology":null},{"url":null,"slug":"vo-q-l-towards-optimal-regret-in-model-free","title":"VO$Q$L: Towards Optimal Regret in Model-free RL with Nonlinear Function Approximation","date":"2022-12-12","arxiv_id":"2212.06069","repositories_listed":0,"syntology":null},{"url":null,"slug":"frugal-reinforcement-based-active-learning","title":"Frugal Reinforcement-based Active Learning","date":"2022-12-09","arxiv_id":"2212.04868","repositories_listed":0,"syntology":null}],"record_sha256":"445871c60c4bd9a8d4f6262c4017882a1627726917e171962adff22cb43fb394","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}