{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/6","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":6,"pages_in_order":18,"rows_per_page":100,"rows":[501,600],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning/papers/5","next":"/method/q-learning/papers/7","papers":[{"paper":null,"slug":"incorporating-deep-q-network-with-multiclass","title":"Comparing Multiclass Classification Algorithms for Financial Distress Prediction","date":"2023-07-08","arxiv_id":"2307.03908","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-value-of-chess-squares","title":"The Value of Chess Squares","date":"2023-07-08","arxiv_id":"2307.05330","n_code_links":0,"syntology":null},{"paper":"/paper/containergym-a-real-world-reinforcement","slug":"containergym-a-real-world-reinforcement","title":"ContainerGym: A Real-World Reinforcement Learning Benchmark for Resource Allocation","date":"2023-07-06","arxiv_id":"2307.02991","n_code_links":1,"syntology":null},{"paper":null,"slug":"offline-reinforcement-learning-with-7","title":"Offline Reinforcement Learning with Imbalanced Datasets","date":"2023-07-06","arxiv_id":"2307.02752","n_code_links":0,"syntology":null},{"paper":null,"slug":"interpretable-and-secure-trajectory","title":"Interpretable and Secure Trajectory Optimization for UAV-Assisted Communication","date":"2023-07-05","arxiv_id":"2307.02002","n_code_links":0,"syntology":null},{"paper":null,"slug":"stability-of-q-learning-through-design-and","title":"Stability of Q-Learning Through Design and Optimism","date":"2023-07-05","arxiv_id":"2307.02632","n_code_links":0,"syntology":null},{"paper":null,"slug":"achieving-stable-training-of-reinforcement","title":"Achieving Stable Training of Reinforcement Learning Agents in Bimodal Environments through Batch Learning","date":"2023-07-03","arxiv_id":"2307.00923","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-risk-sensitive-reinforcement-learning","title":"Is Risk-Sensitive Reinforcement Learning Properly Resolved?","date":"2023-07-02","arxiv_id":"2307.00547","n_code_links":0,"syntology":null},{"paper":"/paper/traceable-group-wise-self-optimizing-feature","slug":"traceable-group-wise-self-optimizing-feature","title":"Traceable Group-Wise Self-Optimizing Feature Transformation Learning: A Dual Optimization Perspective","date":"2023-06-29","arxiv_id":"2306.16893","n_code_links":1,"syntology":null},{"paper":null,"slug":"continuous-time-q-learning-for-mckean-vlasov","title":"Continuous-time q-learning for mean-field control problems","date":"2023-06-28","arxiv_id":"2306.16208","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-reinforcement-learning","title":"Evaluation of Reinforcement Learning Techniques for Trading on a Diverse Portfolio","date":"2023-06-28","arxiv_id":"2309.03202","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-credit-limit-adjustments-under","title":"Optimizing Credit Limit Adjustments Under Adversarial Goals Using Reinforcement Learning","date":"2023-06-27","arxiv_id":"2306.15585","n_code_links":0,"syntology":null},{"paper":null,"slug":"ransomai-ai-powered-ransomware-for-stealthy","title":"RansomAI: AI-powered Ransomware for Stealthy Encryption","date":"2023-06-27","arxiv_id":"2306.15559","n_code_links":0,"syntology":null},{"paper":null,"slug":"decentralized-multi-robot-formation-control","title":"Decentralized Multi-Robot Formation Control Using Reinforcement Learning","date":"2023-06-26","arxiv_id":"2306.14489","n_code_links":0,"syntology":null},{"paper":null,"slug":"action-q-transformer-visual-explanation-in","title":"Action Q-Transformer: Visual Explanation in Deep Reinforcement Learning with Encoder-Decoder Model using Action Query","date":"2023-06-24","arxiv_id":"2306.13879","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-ensemble-q-learning-minimizing-1","title":"Adaptive Ensemble Q-learning: Minimizing Estimation Bias via Error Feedback","date":"2023-06-20","arxiv_id":"2306.11918","n_code_links":0,"syntology":null},{"paper":null,"slug":"autonomous-driving-with-deep-reinforcement","title":"Autonomous Driving with Deep Reinforcement Learning in CARLA Simulation","date":"2023-06-20","arxiv_id":"2306.11217","n_code_links":0,"syntology":null},{"paper":null,"slug":"vanishing-bias-heuristic-guided-reinforcement","title":"Vanishing Bias Heuristic-guided Reinforcement Learning Algorithm","date":"2023-06-17","arxiv_id":"2306.10216","n_code_links":0,"syntology":null},{"paper":null,"slug":"designing-auctions-when-algorithms-learn-to","title":"Algorithmic Collusion in Auctions: Evidence from Controlled Laboratory Experiments","date":"2023-06-15","arxiv_id":"2306.09437","n_code_links":0,"syntology":null},{"paper":"/paper/joint-path-planning-and-power-allocation-of-a","slug":"joint-path-planning-and-power-allocation-of-a","title":"Joint Path planning and Power Allocation of a Cellular-Connected UAV using Apprenticeship Learning via Deep Inverse Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.10071","n_code_links":1,"syntology":null},{"paper":null,"slug":"residual-q-learning-offline-and-online-policy","title":"Residual Q-Learning: Offline and Online Policy Customization without Value","date":"2023-06-15","arxiv_id":"2306.09526","n_code_links":0,"syntology":null},{"paper":null,"slug":"your-room-is-not-private-gradient-inversion","title":"Privacy Risks in Reinforcement Learning for Household Robots","date":"2023-06-15","arxiv_id":"2306.09273","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-based-versus-model-free-feeding-control","title":"Model-based versus model-free feeding control and water quality monitoring for fish growth tracking in aquaculture systems","date":"2023-06-14","arxiv_id":"2306.09915","n_code_links":0,"syntology":null},{"paper":null,"slug":"pruning-the-way-to-reliable-policies-a-multi","title":"Pruning the Way to Reliable Policies: A Multi-Objective Deep Q-Learning Approach to Critical Care","date":"2023-06-13","arxiv_id":"2306.08044","n_code_links":0,"syntology":null},{"paper":null,"slug":"approximate-information-state-based","title":"Approximate information state based convergence analysis of recurrent Q-learning","date":"2023-06-09","arxiv_id":"2306.05991","n_code_links":0,"syntology":null},{"paper":null,"slug":"finite-time-analysis-of-minimax-q-learning","title":"Finite-Time Analysis of Minimax Q-Learning for Two-Player Zero-Sum Markov Games: Switching System Approach","date":"2023-06-09","arxiv_id":"2306.05700","n_code_links":0,"syntology":null},{"paper":null,"slug":"quasi-newton-updating-for-large-scale","title":"Quasi-Newton Updating for Large-Scale Distributed Learning","date":"2023-06-07","arxiv_id":"2306.04111","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-control-of-4","title":"Reinforcement Learning-Based Control of CrazyFlie 2.X Quadrotor","date":"2023-06-06","arxiv_id":"2306.03951","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-q-learning-versus-proximal-policy","title":"Deep Q-Learning versus Proximal Policy Optimization: Performance Comparison in a Material Sorting Task","date":"2023-06-02","arxiv_id":"2306.01451","n_code_links":0,"syntology":null},{"paper":null,"slug":"iql-td-mpc-implicit-q-learning-for","title":"IQL-TD-MPC: Implicit Q-Learning for Hierarchical Model Predictive Control","date":"2023-06-01","arxiv_id":"2306.00867","n_code_links":0,"syntology":null},{"paper":"/paper/off-policy-rl-algorithms-can-be-sample","slug":"off-policy-rl-algorithms-can-be-sample","title":"Off-Policy RL Algorithms Can be Sample-Efficient for Continuous Control via Sample Multiple Reuse","date":"2023-05-29","arxiv_id":"2305.18443","n_code_links":1,"syntology":null},{"paper":null,"slug":"va-learning-as-a-more-efficient-alternative","title":"VA-learning as a more efficient alternative to Q-learning","date":"2023-05-29","arxiv_id":"2305.18161","n_code_links":0,"syntology":null},{"paper":null,"slug":"sample-complexity-of-variance-reduced","title":"Sample Complexity of Variance-reduced Distributionally Robust Q-learning","date":"2023-05-28","arxiv_id":"2305.18420","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-with-reward-machines","title":"Reinforcement Learning With Reward Machines in Stochastic Games","date":"2023-05-27","arxiv_id":"2305.17372","n_code_links":0,"syntology":null},{"paper":null,"slug":"sample-efficient-reinforcement-learning-in-3","title":"Sample Efficient Reinforcement Learning in Mixed Systems through Augmented Samples and Its Applications to Queueing Networks","date":"2023-05-25","arxiv_id":"2305.16483","n_code_links":0,"syntology":null},{"paper":null,"slug":"2305-14656","title":"RSRM: Reinforcement Symbolic Regression Machine","date":"2023-05-24","arxiv_id":"2305.14656","n_code_links":0,"syntology":null},{"paper":"/paper/2305-14550","slug":"2305-14550","title":"When should we prefer Decision Transformers for Offline Reinforcement Learning?","date":"2023-05-23","arxiv_id":"2305.14550","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["prajjwal1/rl_paradigm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-reinforcement-learning-based-multi-1","title":"Deep Reinforcement Learning-based Multi-objective Path Planning on the Off-road Terrain Environment for Ground Vehicles","date":"2023-05-23","arxiv_id":"2305.13783","n_code_links":0,"syntology":null},{"paper":null,"slug":"offline-experience-replay-for-continual","title":"OER: Offline Experience Replay for Continual Offline Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.13804","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-framework-for-provably-stable-and","title":"A Framework for Provably Stable and Consistent Training of Deep Feedforward Networks","date":"2023-05-20","arxiv_id":"2305.12125","n_code_links":0,"syntology":null},{"paper":null,"slug":"bayesian-risk-averse-q-learning-with","title":"Bayesian Risk-Averse Q-Learning with Streaming Observations","date":"2023-05-18","arxiv_id":"2305.11300","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-blessing-of-heterogeneity-in-federated-q","title":"The Blessing of Heterogeneity in Federated Q-Learning: Linear Speedup and Beyond","date":"2023-05-18","arxiv_id":"2305.10697","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-does-agency-impact-human-ai-collaborative","title":"How does agency impact human-AI collaborative design space exploration? A case study on ship design with deep generative models","date":"2023-05-16","arxiv_id":"2305.10451","n_code_links":0,"syntology":null},{"paper":"/paper/an-intelligent-sdwn-routing-algorithm-based","slug":"an-intelligent-sdwn-routing-algorithm-based","title":"An Intelligent SDWN Routing Algorithm Based on Network Situational Awareness and Deep Reinforcement Learning","date":"2023-05-12","arxiv_id":"2305.10441","n_code_links":1,"syntology":null},{"paper":"/paper/mastering-percolation-like-games-with-deep","slug":"mastering-percolation-like-games-with-deep","title":"Mastering Percolation-like Games with Deep Learning","date":"2023-05-12","arxiv_id":"2305.07687","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-practical-robust-reinforcement-learning","title":"On Practical Robust Reinforcement Learning: Practical Uncertainty Set and Double-Agent Algorithm","date":"2023-05-11","arxiv_id":"2305.06657","n_code_links":0,"syntology":null},{"paper":"/paper/extracting-diagnosis-pathways-from-electronic","slug":"extracting-diagnosis-pathways-from-electronic","title":"Extracting Diagnosis Pathways from Electronic Health Records Using Deep Reinforcement Learning","date":"2023-05-10","arxiv_id":"2305.06295","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-position-bias-estimation-against","title":"Position Bias Estimation with Item Embedding for Sparse Dataset","date":"2023-05-10","arxiv_id":"2305.13931","n_code_links":0,"syntology":null},{"paper":"/paper/mixed-integer-optimal-control-via","slug":"mixed-integer-optimal-control-via","title":"Mixed-Integer Optimal Control via Reinforcement Learning: A Case Study on Hybrid Electric Vehicle Energy Management","date":"2023-05-02","arxiv_id":"2305.01461","n_code_links":1,"syntology":null},{"paper":null,"slug":"batch-quantum-reinforcement-learning","title":"BCQQ: Batch-Constraint Quantum Q-Learning with Cyclic Data Re-uploading","date":"2023-04-27","arxiv_id":"2305.00905","n_code_links":0,"syntology":null},{"paper":null,"slug":"safe-q-learning-for-continuous-time-linear","title":"Safe Q-learning for continuous-time linear systems","date":"2023-04-26","arxiv_id":"2304.13573","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-based-equilibria","title":"Learned Collusion","date":"2023-04-25","arxiv_id":"2304.12647","n_code_links":0,"syntology":null},{"paper":"/paper/idql-implicit-q-learning-as-an-actor-critic","slug":"idql-implicit-q-learning-as-an-actor-critic","title":"IDQL: Implicit Q-Learning as an Actor-Critic Method with Diffusion Policies","date":"2023-04-20","arxiv_id":"2304.10573","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["philippe-eecs/idql"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/bridging-rl-theory-and-practice-with-the-1","slug":"bridging-rl-theory-and-practice-with-the-1","title":"Bridging RL Theory and Practice with the Effective Horizon","date":"2023-04-19","arxiv_id":"2304.09853","n_code_links":1,"syntology":null},{"paper":"/paper/h-tsp-hierarchically-solving-the-large-scale","slug":"h-tsp-hierarchically-solving-the-large-scale","title":"H-TSP: Hierarchically Solving the Large-Scale Travelling Salesman Problem","date":"2023-04-19","arxiv_id":"2304.09395","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Learning4Optimization-HUST/H-TSP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-study-on-a-q-learning-algorithm-application","title":"A study on a Q-Learning algorithm application to a manufacturing assembly problem","date":"2023-04-17","arxiv_id":"2304.08375","n_code_links":0,"syntology":null},{"paper":"/paper/collaborative-multi-bs-power-management-for","slug":"collaborative-multi-bs-power-management-for","title":"Collaborative Multi-BS Power Management for Dense Radio Access Network using Deep Reinforcement Learning","date":"2023-04-17","arxiv_id":"2304.07976","n_code_links":1,"syntology":null},{"paper":null,"slug":"robust-decision-making-in-spatial-learning-a","title":"Exploring the Noise Resilience of Successor Features and Predecessor Features Algorithms in One and Two-Dimensional Environments","date":"2023-04-14","arxiv_id":"2304.06894","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-applied-to-an","title":"Deep reinforcement learning applied to an assembly sequence planning problem with user preferences","date":"2023-04-13","arxiv_id":"2304.06567","n_code_links":0,"syntology":null},{"paper":"/paper/automaton-guided-curriculum-generation-for","slug":"automaton-guided-curriculum-generation-for","title":"Automaton-Guided Curriculum Generation for Reinforcement Learning Agents","date":"2023-04-11","arxiv_id":"2304.05271","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-minimum-state","title":"Reinforcement Learning Based Minimum State-flipped Control for the Reachability of Boolean Control Networks","date":"2023-04-11","arxiv_id":"2304.04950","n_code_links":0,"syntology":null},{"paper":null,"slug":"rels-dqn-a-robust-and-efficient-local-search","title":"RELS-DQN: A Robust and Efficient Local Search Framework for Combinatorial Optimization","date":"2023-04-11","arxiv_id":"2304.06048","n_code_links":0,"syntology":null},{"paper":"/paper/generating-a-graph-colouring-heuristic-with","slug":"generating-a-graph-colouring-heuristic-with","title":"Generating a Graph Colouring Heuristic with Deep Q-Learning and Graph Neural Networks","date":"2023-04-08","arxiv_id":"2304.04051","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-optimal-1","title":"Deep Reinforcement Learning Based Optimal Infinite-Horizon Control of Probabilistic Boolean Control Networks","date":"2023-04-07","arxiv_id":"2304.03489","n_code_links":0,"syntology":null},{"paper":null,"slug":"full-gradient-deep-reinforcement-learning-for","title":"Full Gradient Deep Reinforcement Learning for Average-Reward Criterion","date":"2023-04-07","arxiv_id":"2304.03729","n_code_links":0,"syntology":null},{"paper":null,"slug":"computational-role-of-sleep-in-memory","title":"Computational role of sleep in memory reorganization","date":"2023-04-06","arxiv_id":"2304.02873","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-reinforcement-learning","title":"Understanding Reinforcement Learning Algorithms: The Progress from Basic Q-learning to Proximal Policy Optimization","date":"2023-03-31","arxiv_id":"2304.00026","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-learning-based-system-for-path-planning","title":"Q-Learning based system for path planning with unmanned aerial vehicles swarms in obstacle environments","date":"2023-03-30","arxiv_id":"2303.17655","n_code_links":0,"syntology":null},{"paper":"/paper/multi-agent-reinforcement-learning-with-6","slug":"multi-agent-reinforcement-learning-with-6","title":"Multi-Agent Reinforcement Learning with Action Masking for UAV-enabled Mobile Communications","date":"2023-03-29","arxiv_id":"2303.16737","n_code_links":1,"syntology":null},{"paper":null,"slug":"distributed-multi-agent-deep-q-learning-for","title":"Distributed Multi-Agent Deep Q-Learning for Fast Roaming in IEEE 802.11ax Wi-Fi Systems","date":"2023-03-25","arxiv_id":"2304.01210","n_code_links":0,"syntology":null},{"paper":null,"slug":"specific-investments-under-negotiated","title":"Specific investments under negotiated transfer pricing: effects of different surplus sharing parameters on managerial performance: An agent-based simulation with fuzzy Q-learning agents","date":"2023-03-25","arxiv_id":"2303.14515","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-path-following-on-rivers-using","title":"Robust Path Following on Rivers Using Bootstrapped Reinforcement Learning","date":"2023-03-24","arxiv_id":"2303.15178","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-safe-propofol-dosing-during-general","title":"Towards Real-World Applications of Personalized Anesthesia Using Policy Constraint Q Learning for Propofol Infusion Control","date":"2023-03-17","arxiv_id":"2303.10180","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-inspection-method-of-unmanned-aerial","title":"Self-Inspection Method of Unmanned Aerial Vehicles in Power Plants Using Deep Q-Network Reinforcement Learning","date":"2023-03-16","arxiv_id":"2303.09013","n_code_links":0,"syntology":null},{"paper":null,"slug":"smoothed-q-learning","title":"Smoothed Q-learning","date":"2023-03-15","arxiv_id":"2303.08631","n_code_links":0,"syntology":null},{"paper":null,"slug":"recovering-arrhythmic-eeg-transients-from","title":"Recovering Arrhythmic EEG Transients from Their Stochastic Interference","date":"2023-03-14","arxiv_id":"2303.07683","n_code_links":0,"syntology":null},{"paper":null,"slug":"digital-twin-assisted-knowledge-distillation","title":"Digital Twin-Assisted Knowledge Distillation Framework for Heterogeneous Federated Learning","date":"2023-03-10","arxiv_id":"2303.06155","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-framework-for-history-aware-hyperparameter","title":"A Framework for History-Aware Hyperparameter Optimisation in Reinforcement Learning","date":"2023-03-09","arxiv_id":"2303.05186","n_code_links":0,"syntology":null},{"paper":"/paper/cal-ql-calibrated-offline-rl-pre-training-for-1","slug":"cal-ql-calibrated-offline-rl-pre-training-for-1","title":"Cal-QL: Calibrated Offline RL Pre-Training for Efficient Online Fine-Tuning","date":"2023-03-09","arxiv_id":"2303.05479","n_code_links":3,"syntology":null},{"paper":null,"slug":"learning-strategic-value-and-cooperation-in","title":"Learning Strategic Value and Cooperation in Multi-Player Stochastic Games through Side Payments","date":"2023-03-09","arxiv_id":"2303.05307","n_code_links":0,"syntology":null},{"paper":null,"slug":"entropy-environment-transformer-and-offline","title":"Environment Transformer and Policy Optimization for Model-Based Offline Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03811","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploration-via-epistemic-value-estimation","title":"Exploration via Epistemic Value Estimation","date":"2023-03-07","arxiv_id":"2303.04012","n_code_links":0,"syntology":null},{"paper":null,"slug":"double-a3c-deep-reinforcement-learning-on","title":"Double A3C: Deep Reinforcement Learning on OpenAI Gym Games","date":"2023-03-04","arxiv_id":"2303.02271","n_code_links":0,"syntology":null},{"paper":null,"slug":"wasserstein-actor-critic-directed-exploration","title":"Wasserstein Actor-Critic: Directed Exploration via Optimism for Continuous-Actions Control","date":"2023-03-04","arxiv_id":"2303.02378","n_code_links":0,"syntology":null},{"paper":null,"slug":"finite-sample-guarantees-for-nash-q-learning","title":"Finite-sample Guarantees for Nash Q-learning with Linear Function Approximation","date":"2023-03-01","arxiv_id":"2303.00177","n_code_links":0,"syntology":null},{"paper":"/paper/ls-iq-implicit-reward-regularization-for","slug":"ls-iq-implicit-reward-regularization-for","title":"LS-IQ: Implicit Reward Regularization for Inverse Reinforcement Learning","date":"2023-03-01","arxiv_id":"2303.00599","n_code_links":1,"syntology":null},{"paper":null,"slug":"q-cogni-an-integrated-causal-reinforcement","title":"Q-Cogni: An Integrated Causal Reinforcement Learning Framework","date":"2023-02-26","arxiv_id":"2302.13240","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-bellman-s-principle-of-optimality-and","title":"On Bellman's principle of optimality and Reinforcement learning for safety-constrained Markov decision process","date":"2023-02-25","arxiv_id":"2302.13152","n_code_links":0,"syntology":null},{"paper":null,"slug":"provably-efficient-gauss-newton-temporal","title":"Gauss-Newton Temporal Difference Learning with Nonlinear Function Approximation","date":"2023-02-25","arxiv_id":"2302.13087","n_code_links":0,"syntology":null},{"paper":null,"slug":"kernel-based-distributed-q-learning-a","title":"Kernel-Based Distributed Q-Learning: A Scalable Reinforcement Learning Approach for Dynamic Treatment Regimes","date":"2023-02-21","arxiv_id":"2302.10434","n_code_links":0,"syntology":null},{"paper":"/paper/potential-based-reward-shaping-for-learning","slug":"potential-based-reward-shaping-for-learning","title":"Learning to Play Text-based Adventure Games with Maximum Entropy Reinforcement Learning","date":"2023-02-21","arxiv_id":"2302.10720","n_code_links":1,"syntology":null},{"paper":null,"slug":"robust-auto-landing-control-of-an-agile","title":"Robust Auto-landing Control of an agile Regional Jet Using Fuzzy Q-learning","date":"2023-02-21","arxiv_id":"2302.10997","n_code_links":0,"syntology":null},{"paper":null,"slug":"forecasting-and-stabilizing-chaotic-regimes","title":"Forecasting and stabilizing chaotic regimes in two macroeconomic models via artificial intelligence technologies and control methods","date":"2023-02-20","arxiv_id":"2302.12019","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-offline-reinforcement-learning-for-real","title":"Deep Offline Reinforcement Learning for Real-world Treatment Optimization Applications","date":"2023-02-15","arxiv_id":"2302.07549","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-statistical-inference-for-nonlinear","title":"Online Statistical Inference for Nonlinear Stochastic Approximation with Markovian Data","date":"2023-02-15","arxiv_id":"2302.07690","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-lifetime-extended-energy-management","title":"A Lifetime Extended Energy Management Strategy for Fuel Cell Hybrid Electric Vehicles via Self-Learning Fuzzy Reinforcement Learning","date":"2023-02-13","arxiv_id":"2302.06236","n_code_links":0,"syntology":null},{"paper":null,"slug":"computation-offloading-for-uncertain-marine","title":"Computation Offloading for Uncertain Marine Tasks by Cooperation of UAVs and Vessels","date":"2023-02-13","arxiv_id":"2302.06055","n_code_links":0,"syntology":null},{"paper":null,"slug":"differentially-private-deep-q-learning-for","title":"Differentially Private Deep Q-Learning for Pattern Privacy Preservation in MEC Offloading","date":"2023-02-09","arxiv_id":"2302.04608","n_code_links":0,"syntology":null},{"paper":null,"slug":"catch-me-if-you-can-improving-adversaries-in","title":"Catch Me If You Can: Improving Adversaries in Cyber-Security With Q-Learning Algorithms","date":"2023-02-07","arxiv_id":"2302.03768","n_code_links":0,"syntology":null},{"paper":null,"slug":"ensemble-value-functions-for-efficient","title":"Ensemble Value Functions for Efficient Exploration in Multi-Agent Reinforcement Learning","date":"2023-02-07","arxiv_id":"2302.03439","n_code_links":0,"syntology":null}],"record_sha256":"821a2438c63d6bfe4c753ddf0f6adf3d770deedc072c66cbc30c4db6430798e2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}