{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/2","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":18,"rows_per_page":100,"rows":[101,200],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning","next":"/method/q-learning/papers/3","papers":[{"paper":null,"slug":"projection-implicit-q-learning-with-support","title":"Projection Implicit Q-Learning with Support Constraint for Offline Reinforcement Learning","date":"2025-01-15","arxiv_id":"2501.08907","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-inductive-learning-from-answer-sets","title":"Online inductive learning from answer sets for efficient reinforcement learning exploration","date":"2025-01-13","arxiv_id":"2501.07445","n_code_links":0,"syntology":null},{"paper":null,"slug":"perception-guided-eeg-analysis-a-deep","title":"Perception-Guided EEG Analysis: A Deep Learning Approach Inspired by Level of Detail (LOD) Theory","date":"2025-01-11","arxiv_id":"2501.10428","n_code_links":0,"syntology":null},{"paper":null,"slug":"session-level-dynamic-ad-load-optimization","title":"Session-Level Dynamic Ad Load Optimization using Offline Robust Reinforcement Learning","date":"2025-01-09","arxiv_id":"2501.05591","n_code_links":0,"syntology":null},{"paper":null,"slug":"b-dqn-improving-deep-q-learning-by-evolving","title":"$β$-DQN: Improving Deep Q-Learning By Evolving the Behavior","date":"2025-01-01","arxiv_id":"2501.00913","n_code_links":0,"syntology":null},{"paper":null,"slug":"rem-a-scalable-reinforced-multi-expert","title":"REM: A Scalable Reinforced Multi-Expert Framework for Multiplex Influence Maximization","date":"2025-01-01","arxiv_id":"2501.00779","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-optimization-of-storage-systems-using","title":"Dynamic Optimization of Storage Systems Using Reinforcement Learning Techniques","date":"2024-12-29","arxiv_id":"2501.00068","n_code_links":0,"syntology":null},{"paper":null,"slug":"protein-structure-prediction-in-the-3d-hp","title":"Protein Structure Prediction in the 3D HP Model Using Deep Reinforcement Learning","date":"2024-12-29","arxiv_id":"2412.20329","n_code_links":0,"syntology":null},{"paper":"/paper/mobilenetv2-a-lightweight-classification","slug":"mobilenetv2-a-lightweight-classification","title":"MobileNetV2: A lightweight classification model for home-based sleep apnea screening","date":"2024-12-28","arxiv_id":"2412.19967","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-reinforcement-learning-based-task-mapping","title":"A Reinforcement Learning-Based Task Mapping Method to Improve the Reliability of Clustered Manycores","date":"2024-12-26","arxiv_id":"2412.19340","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-based-traffic-aware-base","title":"Deep Learning-Based Traffic-Aware Base Station Sleep Mode and Cell Zooming Strategy in RIS-Aided Multi-Cell Networks","date":"2024-12-25","arxiv_id":"2412.18983","n_code_links":0,"syntology":null},{"paper":null,"slug":"hyperq-opt-q-learning-for-hyperparameter","title":"HyperQ-Opt: Q-learning for Hyperparameter Optimization","date":"2024-12-23","arxiv_id":"2412.17765","n_code_links":0,"syntology":null},{"paper":null,"slug":"acl-ql-adaptive-conservative-level-in-q","title":"ACL-QL: Adaptive Conservative Level in Q-Learning for Offline Reinforcement Learning","date":"2024-12-22","arxiv_id":"2412.16848","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-agent-q-learning-for-real-time-load","title":"Multi-Agent Q-Learning for Real-Time Load Balancing User Association and Handover in Mobile Networks","date":"2024-12-22","arxiv_id":"2412.19835","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-enhancing-network-throughput-using","title":"On Enhancing Network Throughput using Reinforcement Learning in Sliced Testbeds","date":"2024-12-21","arxiv_id":"2412.16673","n_code_links":0,"syntology":null},{"paper":"/paper/decoding-fairness-a-reinforcement-learning","slug":"decoding-fairness-a-reinforcement-learning","title":"Decoding fairness: a reinforcement learning perspective","date":"2024-12-20","arxiv_id":"2412.16249","n_code_links":1,"syntology":null},{"paper":null,"slug":"distribution-free-uncertainty-quantification-2","title":"Distribution-Free Uncertainty Quantification in Mechanical Ventilation Treatment: A Conformal Deep Q-Learning Framework","date":"2024-12-17","arxiv_id":"2412.12597","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-network-driven-reward-prediction-as-a","title":"Neural-Network-Driven Reward Prediction as a Heuristic: Advancing Q-Learning for Mobile Robot Path Planning","date":"2024-12-17","arxiv_id":"2412.12650","n_code_links":0,"syntology":null},{"paper":null,"slug":"integrated-trucks-assignment-and-scheduling","title":"Integrated trucks assignment and scheduling problem with mixed service mode docks: A Q-learning based adaptive large neighborhood search algorithm","date":"2024-12-12","arxiv_id":"2412.09090","n_code_links":0,"syntology":null},{"paper":null,"slug":"pickllm-context-aware-rl-assisted-large","title":"PickLLM: Context-Aware RL-Assisted Large Language Model Routing","date":"2024-12-12","arxiv_id":"2412.12170","n_code_links":0,"syntology":null},{"paper":null,"slug":"edge-delayed-deep-deterministic-policy","title":"Edge Delayed Deep Deterministic Policy Gradient: efficient continuous control for edge scenarios","date":"2024-12-09","arxiv_id":"2412.06390","n_code_links":0,"syntology":null},{"paper":"/paper/drl4aoi-a-drl-framework-for-semantic-aware","slug":"drl4aoi-a-drl-framework-for-semantic-aware","title":"DRL4AOI: A DRL Framework for Semantic-aware AOI Segmentation in Location-Based Services","date":"2024-12-06","arxiv_id":"2412.05437","n_code_links":1,"syntology":null},{"paper":null,"slug":"demonstration-selection-for-in-context","title":"Demonstration Selection for In-Context Learning via Reinforcement Learning","date":"2024-12-05","arxiv_id":"2412.03966","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparative-analysis-of-multi-agent","title":"Comparative Analysis of Multi-Agent Reinforcement Learning Policies for Crop Planning Decision Support","date":"2024-12-03","arxiv_id":"2412.02057","n_code_links":0,"syntology":null},{"paper":null,"slug":"ecg-sleepnet-deep-learning-based","title":"ECG-SleepNet: Deep Learning-Based Comprehensive Sleep Stage Classification Using ECG Signals","date":"2024-12-02","arxiv_id":"2412.01929","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-learning-based-model-free-safety-filter","title":"Q-learning-based Model-free Safety Filter","date":"2024-11-29","arxiv_id":"2411.19809","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-retail-pricing-via-q-learning-a","title":"Dynamic Retail Pricing via Q-Learning -- A Reinforcement Learning Framework for Enhanced Revenue Management","date":"2024-11-27","arxiv_id":"2411.18261","n_code_links":0,"syntology":null},{"paper":"/paper/pretrained-llm-adapted-with-lora-as-a","slug":"pretrained-llm-adapted-with-lora-as-a","title":"Pretrained LLM Adapted with LoRA as a Decision Transformer for Offline RL in Quantitative Trading","date":"2024-11-26","arxiv_id":"2411.17900","n_code_links":1,"syntology":null},{"paper":null,"slug":"time-scale-separation-in-q-learning-extending","title":"Time-Scale Separation in Q-Learning: Extending TD($\\triangle$) for Action-Value Function Decomposition","date":"2024-11-21","arxiv_id":"2411.14019","n_code_links":0,"syntology":null},{"paper":null,"slug":"structure-learning-with-temporal-gaussian","title":"Structure learning with Temporal Gaussian Mixture for model-based Reinforcement Learning","date":"2024-11-18","arxiv_id":"2411.11511","n_code_links":0,"syntology":null},{"paper":null,"slug":"mitigating-relative-over-generalization-in","title":"Mitigating Relative Over-Generalization in Multi-Agent Reinforcement Learning","date":"2024-11-17","arxiv_id":"2411.11099","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcing-competitive-multi-agents-for","title":"Reinforcing Competitive Multi-Agents for Playing So Long Sucker","date":"2024-11-17","arxiv_id":"2411.11057","n_code_links":0,"syntology":null},{"paper":null,"slug":"rationality-based-innate-values-driven","title":"Innate-Values-driven Reinforcement Learning based Cognitive Modeling","date":"2024-11-14","arxiv_id":"2411.09160","n_code_links":0,"syntology":null},{"paper":null,"slug":"coverage-analysis-for-digital-cousin","title":"Coverage Analysis for Digital Cousin Selection -- Improving Multi-Environment Q-Learning","date":"2024-11-13","arxiv_id":"2411.08360","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-robot-assistive-behaviour-with","slug":"enhancing-robot-assistive-behaviour-with","title":"Enhancing Robot Assistive Behaviour with Reinforcement Learning and Theory of Mind","date":"2024-11-11","arxiv_id":"2411.07003","n_code_links":1,"syntology":null},{"paper":null,"slug":"q-sft-q-learning-for-language-models-via","title":"Q-SFT: Q-Learning for Language Models via Supervised Fine-Tuning","date":"2024-11-07","arxiv_id":"2411.05193","n_code_links":0,"syntology":null},{"paper":"/paper/think-smart-act-smarl-analyzing-probabilistic","slug":"think-smart-act-smarl-analyzing-probabilistic","title":"Think Smart, Act SMARL! Analyzing Probabilistic Logic Shields for Multi-Agent Reinforcement Learning","date":"2024-11-07","arxiv_id":"2411.04867","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-deep-reinforcement-2","title":"A Comparative Study of Deep Reinforcement Learning for Crop Production Management","date":"2024-11-06","arxiv_id":"2411.04106","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-the-rainbow-high-performance-deep","slug":"beyond-the-rainbow-high-performance-deep","title":"Beyond The Rainbow: High Performance Deep Reinforcement Learning on a Desktop PC","date":"2024-11-06","arxiv_id":"2411.03820","n_code_links":3,"syntology":{"ran":16,"of":25,"n_ran_checked":13,"n_instrument":3,"unverified":9,"pointer_only":21,"phrase":"16 ran (of which 10 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","official":{"repos":["viptankz/btr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/temporal-difference-learning-using","slug":"temporal-difference-learning-using","title":"Temporal-Difference Learning Using Distributed Error Signals","date":"2024-11-06","arxiv_id":"2411.03604","n_code_links":1,"syntology":{"ran":10,"of":24,"n_ran_checked":10,"n_instrument":0,"unverified":14,"pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 14 unverified","official":{"repos":["social-ai-uoft/ad-paper"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"paper":null,"slug":"dynamic-weight-adjusting-deep-q-networks-for","title":"Dynamic Weight Adjusting Deep Q-Networks for Real-Time Environmental Adaptation","date":"2024-11-04","arxiv_id":"2411.02559","n_code_links":0,"syntology":null},{"paper":"/paper/simulation-of-nanorobots-with-artificial","slug":"simulation-of-nanorobots-with-artificial","title":"Simulation of Nanorobots with Artificial Intelligence and Reinforcement Learning for Advanced Cancer Cell Detection and Tracking","date":"2024-11-04","arxiv_id":"2411.02345","n_code_links":1,"syntology":null},{"paper":null,"slug":"haver-instance-dependent-error-bounds-for","title":"HAVER: Instance-Dependent Error Bounds for Maximum Mean Estimation and Applications to Q-Learning and Monte Carlo Tree Search","date":"2024-11-01","arxiv_id":"2411.00405","n_code_links":0,"syntology":null},{"paper":"/paper/cale-continuous-arcade-learning-environment","slug":"cale-continuous-arcade-learning-environment","title":"CALE: Continuous Arcade Learning Environment","date":"2024-10-31","arxiv_id":"2410.23810","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["farama-foundation/arcade-learning-environment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/q-learning-for-quantile-mdps-a-decomposition","slug":"q-learning-for-quantile-mdps-a-decomposition","title":"Q-learning for Quantile MDPs: A Decomposition, Performance, and Convergence Analysis","date":"2024-10-31","arxiv_id":"2410.24128","n_code_links":1,"syntology":null},{"paper":"/paper/zonal-rl-rrt-integrated-rl-rrt-path-planning","slug":"zonal-rl-rrt-integrated-rl-rrt-path-planning","title":"Zonal RL-RRT: Integrated RL-RRT Path Planning with Collision Probability and Zone Connectivity","date":"2024-10-31","arxiv_id":"2410.24205","n_code_links":1,"syntology":null},{"paper":null,"slug":"refereverything-towards-segmenting-everything","title":"ReferEverything: Towards Segmenting Everything We Can Speak of in Videos","date":"2024-10-30","arxiv_id":"2410.23287","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-driving-car-racing-application-of-deep","title":"Self-Driving Car Racing: Application of Deep Reinforcement Learning","date":"2024-10-30","arxiv_id":"2410.22766","n_code_links":0,"syntology":null},{"paper":"/paper/q-distribution-guided-q-learning-for-offline","slug":"q-distribution-guided-q-learning-for-offline","title":"Q-Distribution guided Q-learning for offline reinforcement learning: Uncertainty penalized Q-value via consistency model","date":"2024-10-27","arxiv_id":"2410.20312","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["evalarzj/qdq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"optimizing-load-scheduling-in-power-grids","title":"Optimizing Load Scheduling in Power Grids Using Reinforcement Learning and Markov Decision Processes","date":"2024-10-23","arxiv_id":"2410.17696","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-novel-reinforcement-learning-model-for-post","title":"A Novel Reinforcement Learning Model for Post-Incident Malware Investigations","date":"2024-10-19","arxiv_id":"2410.15028","n_code_links":0,"syntology":null},{"paper":"/paper/streaming-deep-reinforcement-learning-finally","slug":"streaming-deep-reinforcement-learning-finally","title":"Streaming Deep Reinforcement Learning Finally Works","date":"2024-10-18","arxiv_id":"2410.14606","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mohmdelsayed/streaming-drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"spectrum-sharing-using-deep-reinforcement","title":"Spectrum Sharing using Deep Reinforcement Learning in Vehicular Networks","date":"2024-10-16","arxiv_id":"2410.12521","n_code_links":0,"syntology":null},{"paper":null,"slug":"diar-diffusion-model-guided-implicit-q","title":"DIAR: Diffusion-model-guided Implicit Q-learning with Adaptive Revaluation","date":"2024-10-15","arxiv_id":"2410.11338","n_code_links":0,"syntology":null},{"paper":null,"slug":"diffusion-based-offline-rl-for-improved","title":"Diffusion-Based Offline RL for Improved Decision-Making in Augmented ARC Task","date":"2024-10-15","arxiv_id":"2410.11324","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-agents-with-prioritization-and-1","title":"Learning Agents With Prioritization and Parameter Noise in Continuous State and Action Space","date":"2024-10-15","arxiv_id":"2410.11250","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-objective-optimization-multi-auv","title":"Multi-Objective-Optimization Multi-AUV Assisted Data Collection Framework for IoUT Based on Offline Reinforcement Learning","date":"2024-10-15","arxiv_id":"2410.11282","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-statistical-inference-for-time-varying","title":"Asymptotic Analysis of Sample-averaged Q-learning","date":"2024-10-14","arxiv_id":"2410.10737","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-waveform-selection-for-cognitive-radar","title":"Online waveform selection for cognitive radar","date":"2024-10-14","arxiv_id":"2410.10591","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-llm-ddqn-based-joint-optimization-of","title":"Hybrid LLM-DDQN based Joint Optimization of V2I Communication and Autonomous Driving","date":"2024-10-11","arxiv_id":"2410.08854","n_code_links":0,"syntology":null},{"paper":null,"slug":"gap-dependent-bounds-for-q-learning-using","title":"Gap-Dependent Bounds for Q-Learning using Reference-Advantage Decomposition","date":"2024-10-10","arxiv_id":"2410.07574","n_code_links":0,"syntology":null},{"paper":null,"slug":"uniq-offline-inverse-q-learning-for-avoiding","title":"UNIQ: Offline Inverse Q-learning for Avoiding Undesirable Demonstrations","date":"2024-10-10","arxiv_id":"2410.08307","n_code_links":0,"syntology":null},{"paper":null,"slug":"verifierq-enhancing-llm-test-time-compute","title":"VerifierQ: Enhancing LLM Test Time Compute with Q-Learning-based Verifiers","date":"2024-10-10","arxiv_id":"2410.08048","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-wsl-leveraging-dynamic-programming-for","title":"Q-WSL: Optimizing Goal-Conditioned RL with Weighted Supervised Learning via Dynamic Programming","date":"2024-10-09","arxiv_id":"2410.06648","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-in-complex-action-spaces-without","title":"Learning in complex action spaces without policy gradients","date":"2024-10-08","arxiv_id":"2410.06317","n_code_links":0,"syntology":null},{"paper":null,"slug":"mimicking-human-intuition-cognitive-belief","title":"Mimicking Human Intuition: Cognitive Belief-Driven Q-Learning","date":"2024-10-02","arxiv_id":"2410.01739","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-and-private-marginal-reconstruction","slug":"efficient-and-private-marginal-reconstruction","title":"Efficient and Private Marginal Reconstruction with Local Non-Negativity","date":"2024-10-01","arxiv_id":"2410.01091","n_code_links":1,"syntology":{"ran":17,"of":25,"n_ran_checked":12,"n_instrument":5,"unverified":8,"pointer_only":25,"phrase":"17 ran (of which 2 constructed an object rather than computing a result; 12 with no instrument failure: 10 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 8 unverified","official":{"repos":["bcmullins/efficient-marginal-reconstruction"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":2,"n_ran_no_instrument_failure":12,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"optimizing-photoplethysmography-based-sleep","title":"Optimizing Photoplethysmography-Based Sleep Staging Models by Leveraging Temporal Context for Wearable Devices Applications","date":"2024-10-01","arxiv_id":"2410.00693","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-knowledge-based-multi-objective","title":"Adaptive Knowledge-based Multi-Objective Evolutionary Algorithm for Hybrid Flow Shop Scheduling Problems with Multiple Parallel Batch Processing Stages","date":"2024-09-27","arxiv_id":"2409.18524","n_code_links":0,"syntology":null},{"paper":null,"slug":"experimental-evaluation-of-machine-learning","title":"Experimental Evaluation of Machine Learning Models for Goal-oriented Customer Service Chatbot with Pipeline Architecture","date":"2024-09-27","arxiv_id":"2409.18568","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimized-monte-carlo-tree-search-for","title":"Optimized Monte Carlo Tree Search for Enhanced Decision Making in the FrozenLake Environment","date":"2024-09-25","arxiv_id":"2409.16620","n_code_links":0,"syntology":null},{"paper":"/paper/a-multi-agent-multi-environment-mixed-q","slug":"a-multi-agent-multi-environment-mixed-q","title":"A Multi-Agent Multi-Environment Mixed Q-Learning for Partially Decentralized Wireless Network Optimization","date":"2024-09-24","arxiv_id":"2409.16450","n_code_links":1,"syntology":null},{"paper":null,"slug":"agent-state-based-policies-in-pomdps-beyond","title":"Agent-state based policies in POMDPs: Beyond belief-state MDPs","date":"2024-09-24","arxiv_id":"2409.15703","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-play-video-games-with-intuitive","title":"Learning to Play Video Games with Intuitive Physics Priors","date":"2024-09-20","arxiv_id":"2409.13886","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-efficient-quadratic-q-learning-using","title":"Data-Efficient Quadratic Q-Learning Using LMIs","date":"2024-09-18","arxiv_id":"2409.11986","n_code_links":0,"syntology":null},{"paper":null,"slug":"automating-proton-pbs-treatment-planning-for","title":"Automating proton PBS treatment planning for head and neck cancers using policy gradient-based deep reinforcement learning","date":"2024-09-17","arxiv_id":"2409.11576","n_code_links":0,"syntology":null},{"paper":"/paper/audio-driven-reinforcement-learning-for-head","slug":"audio-driven-reinforcement-learning-for-head","title":"Audio-Driven Reinforcement Learning for Head-Orientation in Naturalistic Environments","date":"2024-09-16","arxiv_id":"2409.10048","n_code_links":1,"syntology":null},{"paper":"/paper/offline-reinforcement-learning-for-learning","slug":"offline-reinforcement-learning-for-learning","title":"Offline Reinforcement Learning for Learning to Dispatch for Job Shop Scheduling","date":"2024-09-16","arxiv_id":"2409.10589","n_code_links":1,"syntology":null},{"paper":null,"slug":"kan-v-s-mlp-for-offline-reinforcement","title":"KAN v.s. MLP for Offline Reinforcement Learning","date":"2024-09-15","arxiv_id":"2409.09653","n_code_links":0,"syntology":null},{"paper":"/paper/learning-discrete-world-models-for-heuristic","slug":"learning-discrete-world-models-for-heuristic","title":"Learning Discrete World Models for Heuristic Search","date":"2024-09-14","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-tracking-a","title":"Deep reinforcement learning for tracking a moving target in jellyfish-like swimming","date":"2024-09-13","arxiv_id":"2409.08815","n_code_links":0,"syntology":null},{"paper":null,"slug":"autonomous-vehicle-decision-making-framework","title":"Autonomous Vehicle Decision-Making Framework for Considering Malicious Behavior at Unsignalized Intersections","date":"2024-09-11","arxiv_id":"2409.17162","n_code_links":0,"syntology":null},{"paper":"/paper/double-successive-over-relaxation-q-learning","slug":"double-successive-over-relaxation-q-learning","title":"Double Successive Over-Relaxation Q-Learning with an Extension to Deep Reinforcement Learning","date":"2024-09-10","arxiv_id":"2409.06356","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-rate-maximization","title":"Reinforcement Learning for Rate Maximization in IRS-aided OWC Networks","date":"2024-09-07","arxiv_id":"2409.04842","n_code_links":0,"syntology":null},{"paper":null,"slug":"reward-directed-score-based-diffusion-models","title":"Reward-Directed Score-Based Diffusion Models via q-Learning","date":"2024-09-07","arxiv_id":"2409.04832","n_code_links":0,"syntology":null},{"paper":null,"slug":"faster-q-learning-algorithms-for-restless","title":"Faster Q-Learning Algorithms for Restless Bandits","date":"2024-09-06","arxiv_id":"2409.05908","n_code_links":0,"syntology":null},{"paper":null,"slug":"whittle-index-learning-algorithms-for","title":"Whittle Index Learning Algorithms for Restless Bandits with Constant Stepsizes","date":"2024-09-06","arxiv_id":"2409.04605","n_code_links":0,"syntology":null},{"paper":"/paper/robust-q-learning-under-corrupted-rewards","slug":"robust-q-learning-under-corrupted-rewards","title":"Robust Q-Learning under Corrupted Rewards","date":"2024-09-05","arxiv_id":"2409.03237","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforcement-learning-enabled-satellite","title":"Reinforcement Learning-enabled Satellite Constellation Reconfiguration and Retasking for Mission-Critical Applications","date":"2024-09-03","arxiv_id":"2409.02270","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerated-multi-objective-task-learning","title":"Accelerated Multi-objective Task Learning using Modified Q-learning Algorithm","date":"2024-09-02","arxiv_id":"2409.01046","n_code_links":0,"syntology":null},{"paper":null,"slug":"stability-of-multiplexed-ncs-based-on-an","title":"Stability of multiplexed NCS based on an epsilon-greedy algorithm for communication selection","date":"2024-09-02","arxiv_id":"2409.00949","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-sample-communication-complexity-trade-off","title":"The Sample-Communication Complexity Trade-off in Federated Q-Learning","date":"2024-08-30","arxiv_id":"2408.16981","n_code_links":0,"syntology":null},{"paper":null,"slug":"coverage-analysis-of-multi-environment-q","title":"Coverage Analysis of Multi-Environment Q-Learning Algorithms for Wireless Network Optimization","date":"2024-08-29","arxiv_id":"2408.16882","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-convergence-of-average-reward-q-learning","title":"On Convergence of Average-Reward Q-Learning in Weakly Communicating Markov Decision Processes","date":"2024-08-29","arxiv_id":"2408.16262","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-td3-for-7-dof-robotic-arm-grasping","title":"Optimizing TD3 for 7-DOF Robotic Arm Grasping: Overcoming Suboptimality with Exploration-Enhanced Contrastive Learning","date":"2024-08-26","arxiv_id":"2408.14009","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-llm-be-a-good-path-planner-based-on","title":"Can LLM be a Good Path Planner based on Prompt Engineering? Mitigating the Hallucination for Path Planning","date":"2024-08-23","arxiv_id":"2408.13184","n_code_links":0,"syntology":null},{"paper":null,"slug":"deviations-from-the-nash-equilibrium-and","title":"Deviations from the Nash equilibrium and emergence of tacit collusion in a two-player optimal execution game with reinforcement learning","date":"2024-08-21","arxiv_id":"2408.11773","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-exploration-in-deep-reinforcement","title":"Efficient Exploration in Deep Reinforcement Learning: A Novel Bayesian Actor-Critic Algorithm","date":"2024-08-19","arxiv_id":"2408.10055","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-conflicts-free-speed-lossless-kan-based","title":"A Conflicts-free, Speed-lossless KAN-based Reinforcement Learning Decision System for Interactive Driving in Roundabouts","date":"2024-08-15","arxiv_id":"2408.08242","n_code_links":0,"syntology":null},{"paper":"/paper/explaining-an-agent-s-future-beliefs-through","slug":"explaining-an-agent-s-future-beliefs-through","title":"Explaining an Agent's Future Beliefs through Temporally Decomposing Future Reward Estimators","date":"2024-08-15","arxiv_id":"2408.08230","n_code_links":1,"syntology":null}],"record_sha256":"dd6594d850906d470b606f027bbe9ce61aa6cdc90e66c04aacdd69f7db56e770","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}