{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/q-learning/papers/9","list_of":"/task/q-learning","task":"Q-Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":20,"rows_per_page":100,"rows":[801,900],"of":1918,"counts":{"archive_papers_tagged":1918,"with_a_code_link":463,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1918,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":102,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":102,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/q-learning","prev":"/task/q-learning/papers/8","next":"/task/q-learning/papers/10","papers":[{"url":null,"slug":"enhanced-q-learning-approach-to-finite-time","title":"Enhanced Q-Learning Approach to Finite-Time Reachability with Maximum Probability for Probabilistic Boolean Control Networks","date":"2023-12-12","arxiv_id":"2312.06904","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-user-association-interference","title":"Joint User Association, Interference Cancellation and Power Control for Multi-IRS Assisted UAV Communications","date":"2023-12-08","arxiv_id":"2312.04786","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-timescale-q-learning-with-function","title":"Two-Timescale Q-Learning with Function Approximation in Zero-Sum Stochastic Games","date":"2023-12-08","arxiv_id":"2312.04905","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-efficient-data-based-off-policy-q-learning","title":"An efficient data-based off-policy Q-learning algorithm for optimal output feedback control of linear systems","date":"2023-12-06","arxiv_id":"2312.03451","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-q-learning-approach-to-the-continuous","title":"A Q-learning approach to the continuous control problem of robot inverted pendulum balancing","date":"2023-12-05","arxiv_id":"2312.02649","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-reinforcement-learning-for-networked","title":"Provable Reinforcement Learning for Networked Control Systems with Stochastic Packet Disordering","date":"2023-12-05","arxiv_id":"2312.02498","repositories_listed":0,"syntology":null},{"url":null,"slug":"spontaneous-coupling-of-q-learning-algorithms","title":"Algorithmic collusion under competitive design","date":"2023-12-05","arxiv_id":"2312.02644","repositories_listed":0,"syntology":null},{"url":null,"slug":"anomaly-detection-via-learning-based","title":"Anomaly Detection via Learning-Based Sequential Controlled Sensing","date":"2023-11-30","arxiv_id":"2312.00088","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-deep-reinforcement-learning-2","title":"Data-efficient Deep Reinforcement Learning for Vehicle Trajectory Control","date":"2023-11-30","arxiv_id":"2311.18393","repositories_listed":0,"syntology":null},{"url":null,"slug":"opensense-an-open-world-sensing-framework-for","title":"OpenSense: An Open-World Sensing Framework for Incremental Learning and Dynamic Sensor Scheduling on Embedded Edge Devices","date":"2023-11-29","arxiv_id":"2311.17358","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-based-optimal-false-data-injection","title":"Q-learning Based Optimal False Data Injection Attack on Probabilistic Boolean Control Networks","date":"2023-11-29","arxiv_id":"2311.17631","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-from-diffusion","title":"Reinforcement Learning from Diffusion Feedback: Q* for Image Search","date":"2023-11-27","arxiv_id":"2311.15648","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-nearly-optimal-and-low-switching-algorithm","title":"A Nearly Optimal and Low-Switching Algorithm for Reinforcement Learning with General Function Approximation","date":"2023-11-26","arxiv_id":"2311.15238","repositories_listed":0,"syntology":null},{"url":null,"slug":"frac-q-learning-a-reinforcement-learning-with","title":"FRAC-Q-Learning: A Reinforcement Learning with Boredom Avoidance Processes for Social Robots","date":"2023-11-26","arxiv_id":"2311.15327","repositories_listed":0,"syntology":null},{"url":null,"slug":"projected-off-policy-q-learning-pop-ql-for","title":"Projected Off-Policy Q-Learning (POP-QL) for Stabilizing Offline Reinforcement Learning","date":"2023-11-25","arxiv_id":"2311.14885","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximation-of-convex-envelope-using","title":"Approximation of Convex Envelope Using Reinforcement Learning","date":"2023-11-24","arxiv_id":"2311.14421","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-open-world-reinforcement-learning","title":"Efficient Open-world Reinforcement Learning via Knowledge Distillation and Autonomous Rule Discovery","date":"2023-11-24","arxiv_id":"2311.14270","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-cooperate-and-communicate-over","title":"Learning to Cooperate and Communicate Over Imperfect Channels","date":"2023-11-24","arxiv_id":"2311.14770","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-optimal-tracking-portfolio-in-incomplete","title":"On optimal tracking portfolio in incomplete markets: The reinforcement learning approach","date":"2023-11-24","arxiv_id":"2311.14318","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-based-decentralized-tdma-for","title":"Machine learning-based decentralized TDMA for VLC IoT networks","date":"2023-11-23","arxiv_id":"2311.14078","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralised-q-learning-for-multi-agent","title":"Decentralised Q-Learning for Multi-Agent Markov Decision Processes with a Satisfiability Criterion","date":"2023-11-21","arxiv_id":"2311.12613","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-for-wireless","title":"Offline Reinforcement Learning for Wireless Network Optimization with Mixture Datasets","date":"2023-11-19","arxiv_id":"2311.11423","repositories_listed":0,"syntology":null},{"url":null,"slug":"genetic-algorithm-enhanced-by-deep","title":"Genetic Algorithm enhanced by Deep Reinforcement Learning in parent selection mechanism and mutation : Minimizing makespan in permutation flow shop scheduling problems","date":"2023-11-10","arxiv_id":"2311.05937","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-algorithmic-trading-a-multi","title":"Advancing Algorithmic Trading: A Multi-Technique Enhancement of Deep Q-Network Models","date":"2023-11-09","arxiv_id":"2311.05743","repositories_listed":0,"syntology":null},{"url":null,"slug":"pointer-networks-with-q-learning-for-op","title":"Pointer Networks with Q-Learning for Combinatorial Optimization","date":"2023-11-05","arxiv_id":"2311.02629","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-for-stochastic-control-under","title":"Q-Learning for Stochastic Control under General Information Structures and Non-Markovian Environments","date":"2023-10-31","arxiv_id":"2311.00123","repositories_listed":0,"syntology":null},{"url":null,"slug":"dgfn-double-generative-flow-networks","title":"DGFN: Double Generative Flow Networks","date":"2023-10-30","arxiv_id":"2310.19685","repositories_listed":0,"syntology":null},{"url":"/paper/weakly-coupled-deep-q-networks","slug":"weakly-coupled-deep-q-networks","title":"Weakly Coupled Deep Q-Networks","date":"2023-10-28","arxiv_id":"2310.18803","repositories_listed":0,"syntology":{"n":10,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":10,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/weakly-coupled-deep-q-networks#ran","syntology_url":"https://syntology.ai/paper/2310.18803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18803"}},"official":null}},{"url":null,"slug":"lifting-the-veil-unlocking-the-power-of-depth","title":"Lifting the Veil: Unlocking the Power of Depth in Q-learning","date":"2023-10-27","arxiv_id":"2310.17915","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-posterior-sampling-via-learning","title":"Model-free Posterior Sampling via Learning Rate Randomization","date":"2023-10-27","arxiv_id":"2310.18186","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrated-freeway-and-arterial-traffic","title":"Integrated Freeway Traffic Control Using Q-Learning with Adjacent Arterial Traffic Considerations","date":"2023-10-25","arxiv_id":"2310.16748","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-convergence-and-sample-complexity","title":"On the Convergence and Sample Complexity Analysis of Deep Q-Networks with $ε$-Greedy Exploration","date":"2023-10-24","arxiv_id":"2310.16173","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-local-path","title":"Reinforcement learning based local path planning for mobile robot","date":"2023-10-24","arxiv_id":"2403.12463","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-on-the-water-applying-drl-to-autonomous","title":"AI on the Water: Applying DRL to Autonomous Vessel Navigation","date":"2023-10-23","arxiv_id":"2310.14938","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-are-bandits-robust-to-misspecification","title":"Bad Values but Good Behavior: Learning Highly Misspecified Bandits and MDPs","date":"2023-10-13","arxiv_id":"2310.09358","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrated-sensing-and-communication-neighbor","title":"Integrated Sensing and Communication Neighbor Discovery for MANET with Gossip Mechanism","date":"2023-10-11","arxiv_id":"2310.07292","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-factorized-q-learning-for-cooperative","title":"Inverse Factorized Q-Learning for Cooperative Multi-agent Imitation Learning","date":"2023-10-10","arxiv_id":"2310.06801","repositories_listed":0,"syntology":null},{"url":null,"slug":"suppressing-overestimation-in-q-learning","title":"Suppressing Overestimation in Q-Learning through Adversarial Behaviors","date":"2023-10-10","arxiv_id":"2310.06286","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-value-alignment-through-preference","title":"Dynamic value alignment through preference aggregation of multiple objectives","date":"2023-10-09","arxiv_id":"2310.05871","repositories_listed":0,"syntology":null},{"url":null,"slug":"diff-transfer-model-based-robotic","title":"Diff-Transfer: Model-based Robotic Manipulation Skill Transfer via Differentiable Physics Simulation","date":"2023-10-07","arxiv_id":"2310.04930","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-twin-assisted-deep-reinforcement","title":"Digital Twin Assisted Deep Reinforcement Learning for Online Admission Control in Sliced Network","date":"2023-10-07","arxiv_id":"2310.09299","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-reinforcement-learning-to-option","title":"Applying Reinforcement Learning to Option Pricing and Hedging","date":"2023-10-06","arxiv_id":"2310.04336","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-control-of-district-cooling-energy","title":"Optimal Control of District Cooling Energy Plant with Reinforcement Learning and MPC","date":"2023-10-05","arxiv_id":"2310.03814","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-15","title":"A Deep Reinforcement Learning Approach for Interactive Search with Sentence-level Feedback","date":"2023-10-03","arxiv_id":"2310.03043","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-analysis-of-whittle-index-based-q","title":"Finite-Time Analysis of Whittle Index based Q-Learning for Restless Multi-Armed Bandits with Neural Network Function Approximation","date":"2023-10-03","arxiv_id":"2310.02147","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-reinforcement-learning-to-optimize","title":"Using Reinforcement Learning to Optimize Responses in Care Processes: A Case Study on Aggression Incidents","date":"2023-10-02","arxiv_id":"2310.00981","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-adaptive-fuzzy","title":"Reinforcement learning adaptive fuzzy controller for lighting systems: application to aircraft cabin","date":"2023-09-30","arxiv_id":"2310.00525","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-bellman-operator-for-convergence-of-q","title":"Multi-Bellman operator for convergence of $Q$-learning with linear function approximation","date":"2023-09-28","arxiv_id":"2309.16819","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-trust-a-reinforcement-learning","title":"Decoding trust: A reinforcement learning perspective","date":"2023-09-26","arxiv_id":"2309.14598","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-double-q-learning-for-continuous","title":"Adapting Double Q-Learning for Continuous Reinforcement Learning","date":"2023-09-25","arxiv_id":"2309.14471","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-swarm-deployment-and-trajectory-for-3d","title":"UAV Swarm Deployment and Trajectory for 3D Area Coverage via Reinforcement Learning","date":"2023-09-21","arxiv_id":"2309.11992","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-driven-patient-monitoring-with-multi-agent","title":"Adaptive Multi-Agent Deep Reinforcement Learning for Timely Healthcare Interventions","date":"2023-09-20","arxiv_id":"2309.10980","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-quantum-architecture-search-1","title":"Differentiable Quantum Architecture Search for Quantum Reinforcement Learning","date":"2023-09-19","arxiv_id":"2309.10392","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-deep-q-learning-based-path-selection","title":"Double Deep Q-Learning-based Path Selection and Service Placement for Latency-Sensitive Beyond 5G Applications","date":"2023-09-18","arxiv_id":"2309.10180","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-transformer-scalable-offline-reinforcement","title":"Q-Transformer: Scalable Offline Reinforcement Learning via Autoregressive Q-Functions","date":"2023-09-18","arxiv_id":"2309.10150","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-sustaining-multiple-access-with-1","title":"Self-Sustaining Multiple Access with Continual Deep Reinforcement Learning for Dynamic Metaverse Applications","date":"2023-09-18","arxiv_id":"2309.10177","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-h-infinity-control-with-a-real","title":"Data-Driven H-infinity Control with a Real-Time and Efficient Reinforcement Learning Algorithm: An Application to Autonomous Mobility-on-Demand Systems","date":"2023-09-16","arxiv_id":"2309.08880","repositories_listed":0,"syntology":null},{"url":null,"slug":"harnessing-deep-q-learning-for-enhanced","title":"Harnessing Deep Q-Learning for Enhanced Statistical Arbitrage in High-Frequency Trading: A Comprehensive Exploration","date":"2023-09-13","arxiv_id":"2311.10718","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-q-learning-approach-for-adherence-aware","title":"A Q-learning Approach for Adherence-Aware Recommendations","date":"2023-09-12","arxiv_id":"2309.06519","repositories_listed":0,"syntology":null},{"url":null,"slug":"career-path-recommendations-for-long-term","title":"Career Path Recommendations for Long-term Income Maximization: A Reinforcement Learning Approach","date":"2023-09-11","arxiv_id":"2309.05391","repositories_listed":0,"syntology":null},{"url":null,"slug":"convex-q-learning-in-a-stochastic-environment","title":"Convex Q Learning in a Stochastic Environment: Extended Version","date":"2023-09-10","arxiv_id":"2309.05105","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deeprl-based-joint-power-and","title":"Multi Agent DeepRL based Joint Power and Subchannel Allocation in IAB networks","date":"2023-08-31","arxiv_id":"2309.00144","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-based-trajectory-design-for-cellular","title":"Physics-Based Trajectory Design for Cellular-Connected UAV in Rainy Environments Based on Deep Reinforcement Learning","date":"2023-08-31","arxiv_id":"2309.00017","repositories_listed":0,"syntology":null},{"url":null,"slug":"actuator-trajectory-planning-for-uavs-with","title":"Actuator Trajectory Planning for UAVs with Overhead Manipulator using Reinforcement Learning","date":"2023-08-24","arxiv_id":"2308.12843","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-algorithm-with-improved-sample","title":"Improving Sample Efficiency of Model-Free Algorithms for Zero-Sum Markov Games","date":"2023-08-17","arxiv_id":"2308.08858","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-battery-management","title":"Reinforcement Learning for Battery Management in Dairy Farming","date":"2023-08-17","arxiv_id":"2308.09023","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-rl-augmented-cold","title":"On-demand Cold Start Frequency Reduction with Off-Policy Reinforcement Learning in Serverless Computing","date":"2023-08-15","arxiv_id":"2308.07541","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-classical-and-deep","title":"A Comparison of Classical and Deep Reinforcement Learning Methods for HVAC Control","date":"2023-08-10","arxiv_id":"2308.05711","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-decentralized-q-learning-two","title":"Unsynchronized Decentralized Q-Learning: Two Timescale Analysis By Persistence","date":"2023-08-07","arxiv_id":"2308.03239","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-network-for-stochastic-process","title":"Deep Q-Network for Stochastic Process Environments","date":"2023-08-07","arxiv_id":"2308.03316","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-optimal-q-learning-with-nearest","title":"Minimax Optimal Q Learning with Nearest Neighbors","date":"2023-08-03","arxiv_id":"2308.01490","repositories_listed":0,"syntology":null},{"url":null,"slug":"stability-of-multi-agent-learning-convergence","title":"Stability of Multi-Agent Learning: Convergence in Network Games with Many Players","date":"2023-07-26","arxiv_id":"2307.13922","repositories_listed":0,"syntology":null},{"url":"/paper/parallel-q-learning-scaling-off-policy","slug":"parallel-q-learning-scaling-off-policy","title":"Parallel $Q$-Learning: Scaling Off-policy Reinforcement Learning under Massively Parallel Simulation","date":"2023-07-24","arxiv_id":"2307.12983","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parallel-q-learning-scaling-off-policy#ran","syntology_url":"https://syntology.ai/paper/2307.12983","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12983"}},"official":null}},{"url":null,"slug":"adversarial-agents-for-attacking-inaudible","title":"Adversarial Agents For Attacking Inaudible Voice Activated Devices","date":"2023-07-23","arxiv_id":"2307.12204","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-flexible-framework-for-incorporating","title":"A Flexible Framework for Incorporating Patient Preferences Into Q-Learning","date":"2023-07-22","arxiv_id":"2307.12022","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-3d-beam-reforming-for-hovering","title":"Distributed 3D-Beam Reforming for Hovering-Tolerant UAVs Communication over Coexistence: A Deep-Q Learning for Intelligent Space-Air-Ground Integrated Networks","date":"2023-07-18","arxiv_id":"2307.09325","repositories_listed":0,"syntology":null},{"url":null,"slug":"credit-assignment-challenges-and","title":"Credit Assignment: Challenges and Opportunities in Developing Human-like AI Agents","date":"2023-07-16","arxiv_id":"2307.08171","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-the-dynamic","title":"Deep reinforcement learning for the dynamic vehicle dispatching problem: An event-based approach","date":"2023-07-13","arxiv_id":"2307.07508","repositories_listed":0,"syntology":null},{"url":null,"slug":"realtime-spectrum-monitoring-via","title":"Realtime Spectrum Monitoring via Reinforcement Learning -- A Comparison Between Q-Learning and Heuristic Methods","date":"2023-07-11","arxiv_id":"2307.05763","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-edge-of-stability","title":"Investigating the Edge of Stability Phenomenon in Reinforcement Learning","date":"2023-07-09","arxiv_id":"2307.04210","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-value-of-chess-squares","title":"The Value of Chess Squares","date":"2023-07-08","arxiv_id":"2307.05330","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-7","title":"Offline Reinforcement Learning with Imbalanced Datasets","date":"2023-07-06","arxiv_id":"2307.02752","repositories_listed":0,"syntology":null},{"url":null,"slug":"elastic-decision-transformer-1","title":"Elastic Decision Transformer","date":"2023-07-05","arxiv_id":"2307.02484","repositories_listed":0,"syntology":null},{"url":null,"slug":"llql-logistic-likelihood-q-learning-for","title":"LLQL: Logistic Likelihood Q-Learning for Reinforcement Learning","date":"2023-07-05","arxiv_id":"2307.02345","repositories_listed":0,"syntology":null},{"url":null,"slug":"stability-of-q-learning-through-design-and","title":"Stability of Q-Learning Through Design and Optimism","date":"2023-07-05","arxiv_id":"2307.02632","repositories_listed":0,"syntology":null},{"url":null,"slug":"achieving-stable-training-of-reinforcement","title":"Achieving Stable Training of Reinforcement Learning Agents in Bimodal Environments through Batch Learning","date":"2023-07-03","arxiv_id":"2307.00923","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-risk-sensitive-reinforcement-learning","title":"Is Risk-Sensitive Reinforcement Learning Properly Resolved?","date":"2023-07-02","arxiv_id":"2307.00547","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-time-q-learning-for-mckean-vlasov","title":"Continuous-time q-learning for mean-field control problems","date":"2023-06-28","arxiv_id":"2306.16208","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-reinforcement-learning","title":"Evaluation of Reinforcement Learning Techniques for Trading on a Diverse Portfolio","date":"2023-06-28","arxiv_id":"2309.03202","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-credit-limit-adjustments-under","title":"Optimizing Credit Limit Adjustments Under Adversarial Goals Using Reinforcement Learning","date":"2023-06-27","arxiv_id":"2306.15585","repositories_listed":0,"syntology":null},{"url":null,"slug":"ransomai-ai-powered-ransomware-for-stealthy","title":"RansomAI: AI-powered Ransomware for Stealthy Encryption","date":"2023-06-27","arxiv_id":"2306.15559","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-multi-robot-formation-control","title":"Decentralized Multi-Robot Formation Control Using Reinforcement Learning","date":"2023-06-26","arxiv_id":"2306.14489","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-q-transformer-visual-explanation-in","title":"Action Q-Transformer: Visual Explanation in Deep Reinforcement Learning with Encoder-Decoder Model using Action Query","date":"2023-06-24","arxiv_id":"2306.13879","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-ensemble-q-learning-minimizing-1","title":"Adaptive Ensemble Q-learning: Minimizing Estimation Bias via Error Feedback","date":"2023-06-20","arxiv_id":"2306.11918","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-driving-with-deep-reinforcement","title":"Autonomous Driving with Deep Reinforcement Learning in CARLA Simulation","date":"2023-06-20","arxiv_id":"2306.11217","repositories_listed":0,"syntology":null},{"url":null,"slug":"vanishing-bias-heuristic-guided-reinforcement","title":"Vanishing Bias Heuristic-guided Reinforcement Learning Algorithm","date":"2023-06-17","arxiv_id":"2306.10216","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-auctions-when-algorithms-learn-to","title":"Algorithmic Collusion in Auctions: Evidence from Controlled Laboratory Experiments","date":"2023-06-15","arxiv_id":"2306.09437","repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-q-learning-offline-and-online-policy","title":"Residual Q-Learning: Offline and Online Policy Customization without Value","date":"2023-06-15","arxiv_id":"2306.09526","repositories_listed":0,"syntology":null},{"url":null,"slug":"your-room-is-not-private-gradient-inversion","title":"Privacy Risks in Reinforcement Learning for Household Robots","date":"2023-06-15","arxiv_id":"2306.09273","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-versus-model-free-feeding-control","title":"Model-based versus model-free feeding control and water quality monitoring for fish growth tracking in aquaculture systems","date":"2023-06-14","arxiv_id":"2306.09915","repositories_listed":0,"syntology":null}],"record_sha256":"4d442e8c4725ac67c289ec5414c791a8fbfd2ab511f7586fd201125a9bd287f4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}