{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/q-learning/papers/14","list_of":"/task/q-learning","task":"Q-Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":14,"pages_in_order":20,"rows_per_page":100,"rows":[1301,1400],"of":1918,"counts":{"archive_papers_tagged":1918,"with_a_code_link":463,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1918,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":102,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":102,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/q-learning","prev":"/task/q-learning/papers/13","next":"/task/q-learning/papers/15","papers":[{"url":null,"slug":"reinforcement-learning-for-traffic-signal","title":"Reinforcement Learning for Traffic Signal Control: Comparison with Commercial Systems","date":"2021-04-21","arxiv_id":"2104.10455","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simulated-experiment-to-explore-robotic","title":"A Simulated Experiment to Explore Robotic Dialogue Strategies for People with Dementia","date":"2021-04-18","arxiv_id":"2104.08940","repositories_listed":0,"syntology":null},{"url":null,"slug":"actionable-models-unsupervised-offline","title":"Actionable Models: Unsupervised Offline Reinforcement Learning of Robotic Skills","date":"2021-04-15","arxiv_id":"2104.07749","repositories_listed":0,"syntology":null},{"url":null,"slug":"prospect-theoretic-q-learning","title":"Prospect-theoretic Q-learning","date":"2021-04-12","arxiv_id":"2104.05311","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-resilience-for-multi-agent-qd","title":"Towards Resilience for Multi-Agent $QD$-Learning","date":"2021-04-07","arxiv_id":"2104.03153","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-deep-reinforcement-learning-for-3","title":"Distributed Deep Reinforcement Learning for Collaborative Spectrum Sharing","date":"2021-04-06","arxiv_id":"2104.02059","repositories_listed":0,"syntology":null},{"url":null,"slug":"solo-search-online-learn-offline-for","title":"SOLO: Search Online, Learn Offline for Combinatorial Optimization Problems","date":"2021-04-04","arxiv_id":"2104.01646","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-double-deep-q-learning-for-joint","title":"Federated Double Deep Q-learning for Joint Delay and Energy Minimization in IoT networks","date":"2021-04-02","arxiv_id":"2104.11320","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-of-finite-memory-q-learning-for","title":"Convergence of Finite Memory Q-Learning for POMDPs and Near Optimality of Learned Policies under Filter Stability","date":"2021-03-22","arxiv_id":"2103.12158","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-on-scenario-tree","title":"Reinforcement Learning based on Scenario-tree MPC for ASVs","date":"2021-03-22","arxiv_id":"2103.11949","repositories_listed":0,"syntology":null},{"url":null,"slug":"softmax-with-regularization-better-value","title":"Regularized Softmax Deep Multi-Agent $Q$-Learning","date":"2021-03-22","arxiv_id":"2103.11883","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-quantum-compiling-with-double-q","title":"Variational quantum compiling with double Q-learning","date":"2021-03-22","arxiv_id":"2103.11611","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-jointly-optimal-design-of-control-and","title":"A Jointly Optimal Design of Control and Scheduling in Networked Systems under Denial-of-Service Attacks","date":"2021-03-10","arxiv_id":"2103.05893","repositories_listed":0,"syntology":null},{"url":null,"slug":"s4rl-surprisingly-simple-self-supervision-for","title":"S4RL: Surprisingly Simple Self-Supervision for Offline Reinforcement Learning","date":"2021-03-10","arxiv_id":"2103.06326","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-effect-of-q-function-reuse-on-the-total","title":"The Effect of Q-function Reuse on the Total Regret of Tabular, Model-Free, Reinforcement Learning","date":"2021-03-07","arxiv_id":"2103.04416","repositories_listed":0,"syntology":null},{"url":null,"slug":"correlated-deep-q-learning-based-microgrid","title":"Correlated Deep Q-learning based Microgrid Energy Management","date":"2021-03-06","arxiv_id":"2103.04152","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-microgrid-energy-management-a","title":"Decentralized Microgrid Energy Management: A Multi-agent Correlated Q-learning Approach","date":"2021-03-06","arxiv_id":"2103.04154","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-bootstrapping-for-q-learning","title":"Ensemble Bootstrapping for Q-Learning","date":"2021-02-28","arxiv_id":"2103.00445","repositories_listed":0,"syntology":null},{"url":null,"slug":"potential-impacts-of-smart-homes-on-human","title":"Potential Impacts of Smart Homes on Human Behavior: A Reinforcement Learning Approach","date":"2021-02-26","arxiv_id":"2102.13307","repositories_listed":0,"syntology":null},{"url":null,"slug":"no-regret-reinforcement-learning-with-heavy","title":"No-Regret Reinforcement Learning with Heavy-Tailed Rewards","date":"2021-02-25","arxiv_id":"2102.12769","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approach-for-resource","title":"Reinforcement learning approach for resource allocation in humanitarian logistics","date":"2021-02-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-learning-based-iaas-composition","title":"Sequential Learning-based IaaS Composition","date":"2021-02-24","arxiv_id":"2102.12598","repositories_listed":0,"syntology":null},{"url":null,"slug":"greedy-multi-step-off-policy-reinforcement-1","title":"Greedy-Step Off-Policy Reinforcement Learning","date":"2021-02-23","arxiv_id":"2102.11717","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-analysis-of-asynchronous-q","title":"A Discrete-Time Switching System Analysis of Q-learning","date":"2021-02-17","arxiv_id":"2102.08583","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperation-and-reputation-dynamics-with","title":"Cooperation and Reputation Dynamics with Reinforcement Learning","date":"2021-02-15","arxiv_id":"2102.07523","repositories_listed":0,"syntology":null},{"url":null,"slug":"reversible-action-design-for-combinatorial","title":"Reversible Action Design for Combinatorial Optimization with Reinforcement Learning","date":"2021-02-14","arxiv_id":"2102.07210","repositories_listed":0,"syntology":null},{"url":null,"slug":"tightening-the-dependence-on-horizon-in-the","title":"Is Q-Learning Minimax Optimal? A Tight Sample Complexity Analysis","date":"2021-02-12","arxiv_id":"2102.06548","repositories_listed":0,"syntology":null},{"url":null,"slug":"hedging-of-financial-derivative-contracts-via","title":"Hedging of Financial Derivative Contracts via Monte Carlo Tree Search","date":"2021-02-11","arxiv_id":"2102.06274","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-agent-complex-environment-efficient","title":"Simple Agent, Complex Environment: Efficient Reinforcement Learning with Agent States","date":"2021-02-10","arxiv_id":"2102.05261","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-augmented-q-learning","title":"Model-Augmented Q-learning","date":"2021-02-07","arxiv_id":"2102.03866","repositories_listed":0,"syntology":null},{"url":null,"slug":"experience-based-heuristic-search-robust","title":"Experience-Based Heuristic Search: Robust Motion Planning with Deep Q-Learning","date":"2021-02-05","arxiv_id":"2102.03127","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-motion-planning-algorithms-for","title":"A review of motion planning algorithms for intelligent robotics","date":"2021-02-04","arxiv_id":"2102.02376","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-image-1","title":"Deep reinforcement learning-based image classification achieves perfect testing set accuracy for MRI brain tumors with a training set of only 30 images","date":"2021-02-04","arxiv_id":"2102.02895","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lyapunov-theory-for-finite-sample","title":"A Lyapunov Theory for Finite-Sample Guarantees of Asynchronous Q-Learning and TD-Learning Variants","date":"2021-02-02","arxiv_id":"2102.01567","repositories_listed":0,"syntology":null},{"url":null,"slug":"qos-aware-power-minimization-of-distributed","title":"QoS-Aware Power Minimization of Distributed Many-Core Servers using Transfer Q-Learning","date":"2021-02-02","arxiv_id":"2102.01348","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-reinforcement-learning-de-novo","title":"A step toward a reinforcement learning de novo genome assembler","date":"2021-02-02","arxiv_id":"2102.02649","repositories_listed":0,"syntology":null},{"url":null,"slug":"coordiq-coordinated-q-learning-for-electric","title":"CoordiQ : Coordinated Q-learning for Electric Vehicle Charging Recommendation","date":"2021-01-28","arxiv_id":"2102.00847","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-per-antenna","title":"Reinforcement Learning based Per-antenna Discrete Power Control for Massive MIMO Systems","date":"2021-01-28","arxiv_id":"2101.12154","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-assisted-beamforming","title":"Reinforcement Learning Assisted Beamforming for Inter-cell Interference Mitigation in 5G Massive MIMO Networks","date":"2021-01-27","arxiv_id":"2103.11782","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-android-malware-detection-system","title":"Robust Android Malware Detection System against Adversarial Attacks using Q-Learning","date":"2021-01-27","arxiv_id":"2101.12031","repositories_listed":0,"syntology":null},{"url":null,"slug":"channel-estimation-via-successive-denoising","title":"Channel Estimation via Successive Denoising in MIMO OFDM Systems: A Reinforcement Learning Approach","date":"2021-01-25","arxiv_id":"2101.10300","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-optimal-stopping-problems-with-deep-q","title":"Solving optimal stopping problems with Deep Q-Learning","date":"2021-01-24","arxiv_id":"2101.09682","repositories_listed":0,"syntology":null},{"url":null,"slug":"fire-threat-detection-from-videos-with-q","title":"Fire Threat Detection From Videos with Q-Rough Sets","date":"2021-01-21","arxiv_id":"2101.08459","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-recommender-1","title":"Reinforcement learning based recommender systems: A survey","date":"2021-01-15","arxiv_id":"2101.06286","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-augmented-index-policy-for-optimal","title":"Learning Augmented Index Policy for Optimal Service Placement at the Network Edge","date":"2021-01-10","arxiv_id":"2101.03641","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-and-scalable-routing-with-multi-agent","title":"Robust and Scalable Routing with Multi-Agent Deep Reinforcement Learning for MANETs","date":"2021-01-09","arxiv_id":"2101.03273","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-coupled-deep-q-learning-for","title":"Safe Coupled Deep Q-Learning for Recommendation Systems","date":"2021-01-08","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-distribution-shift-in-online","title":"Addressing Distribution Shift in Online Reinforcement Learning with Offline Datasets","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-learning-from-dynamic-demonstration","title":"Deep Q Learning from Dynamic Demonstration with Behavioral Cloning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-learning-with-low-switching-cost","title":"Deep Q-Learning with Low Switching Cost","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-anti","title":"Deep Reinforcement Learning-based Anti-jamming Power Allocation in a Two-cell NOMA Network","date":"2021-01-01","arxiv_id":"2101.00270","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-q-learning-new-analysis-and-sharper","title":"Double Q-learning: New Analysis and Sharper Finite-time Bound","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-movement-strategies-for-moving","title":"Learning Movement Strategies for Moving Target Defense","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"optimistic-exploration-with-backward","title":"Optimistic Exploration with Backward Bootstrapped Bonus for Deep Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"preventing-value-function-collapse-in","title":"Preventing Value Function Collapse in Ensemble Q-Learning by Maximizing Representation Diversity","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"success-rate-targeted-reinforcement-learning","title":"Success-Rate Targeted Reinforcement Learning by Disorientation Penalty","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-weighted-offline-reinforcement","title":"Uncertainty Weighted Offline Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"weighted-bellman-backups-for-improved-signal","title":"Weighted Bellman Backups for Improved Signal-to-Noise in Q-Updates","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"blackwell-online-learning-for-markov-decision","title":"Blackwell Online Learning for Markov Decision Processes","date":"2020-12-28","arxiv_id":"2012.14043","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-planning-and-control-in-vision","title":"Disentangled Planning and Control in Vision Based Robotics via Reward Machines","date":"2020-12-28","arxiv_id":"2012.14464","repositories_listed":0,"syntology":null},{"url":null,"slug":"assured-rl-reinforcement-learning-with-almost","title":"Assured RL: Reinforcement Learning with Almost Sure Constraints","date":"2020-12-24","arxiv_id":"2012.13036","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-q-learning-with-state-tracking","title":"Distributed Q-Learning with State Tracking for Multi-agent Networked Control","date":"2020-12-22","arxiv_id":"2012.12383","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-reasoning-by-selecting-subgoals-with","title":"Goal Reasoning by Selecting Subgoals with Deep Q-Learning","date":"2020-12-22","arxiv_id":"2012.12335","repositories_listed":0,"syntology":null},{"url":null,"slug":"stabilizing-q-learning-via-soft-mellowmax","title":"Stabilizing Q Learning Via Soft Mellowmax Operator","date":"2020-12-17","arxiv_id":"2012.09456","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-via-1","title":"Sample-Efficient Reinforcement Learning via Counterfactual-Based Data Augmentation","date":"2020-12-16","arxiv_id":"2012.09092","repositories_listed":0,"syntology":null},{"url":null,"slug":"deploying-reinforcement-learning-in-water","title":"Deploying Reinforcement Learning in Water Transport","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"virtual-autonomous-driving-with-reinforcement","title":"Virtual Autonomous Driving with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-off-policy-reinforcement","title":"Semi-Supervised Off Policy Reinforcement Learning","date":"2020-12-09","arxiv_id":"2012.04809","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-pseudo-labeling-with-reinforcement","title":"Selective Pseudo-Labeling with Reinforcement Learning for Semi-Supervised Domain Adaptation","date":"2020-12-07","arxiv_id":"2012.03438","repositories_listed":0,"syntology":null},{"url":null,"slug":"amortized-q-learning-with-model-based-action","title":"Amortized Q-learning with Model-based Action Proposals for Autonomous Driving on Highways","date":"2020-12-06","arxiv_id":"2012.03234","repositories_listed":0,"syntology":null},{"url":null,"slug":"hippocampal-representations-emerge-when-1","title":"Hippocampal representations emerge when training recurrent neural networks on a memory dependent maze navigation task","date":"2020-12-02","arxiv_id":"2012.01328","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-correcting-q-learning","title":"Self-correcting Q-Learning","date":"2020-12-02","arxiv_id":"2012.01100","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-convergent-variant-of-q-learning-with","title":"A new convergent variant of Q-learning with linear function approximation","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-switching-system-perspective-and-1","title":"A Unified Switching System Perspective and Convergence Analysis of Q-Learning Algorithms","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"agnostic-q-learning-with-function-1","title":"Agnostic $Q$-learning with Function Approximation in Deterministic Systems: Near-Optimal Bounds on Approximation Error and Sample Complexity","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"can-temporal-difference-and-q-learning-learn-1","title":"Can Temporal-Diﬀerence and Q-Learning Learn Representation? A Mean-Field Theory","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-multi-agent-reinforcement-learning-1","title":"Robust Multi-Agent Reinforcement Learning with Model Uncertainty","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-a-particle","title":"Deep reinforcement learning with a particle dynamics environment applied to emergency evacuation of a room with obstacles","date":"2020-11-30","arxiv_id":"2012.00065","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-active-vision-for-a-humanoid-soccer","title":"Real-time Active Vision for a Humanoid Soccer Robot Using Deep Reinforcement Learning","date":"2020-11-27","arxiv_id":"2011.13851","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-joint-path-and","title":"Reinforcement Learning-based Joint Path and Energy Optimization of Cellular-Connected Unmanned Aerial Vehicles","date":"2020-11-27","arxiv_id":"2011.13744","repositories_listed":0,"syntology":null},{"url":null,"slug":"diluted-near-optimal-expert-demonstrations","title":"Diluted Near-Optimal Expert Demonstrations for Guiding Dialogue Stochastic Policy Optimisation","date":"2020-11-25","arxiv_id":"2012.04687","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-5","title":"Multi-Agent Reinforcement Learning for Markov Routing Games: A New Modeling Paradigm For Dynamic Traffic Assignment","date":"2020-11-22","arxiv_id":"2011.10915","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-multi-objective-reinforcement","title":"Provable Multi-Objective Reinforcement Learning with Generative Models","date":"2020-11-19","arxiv_id":"2011.10134","repositories_listed":0,"syntology":null},{"url":null,"slug":"c-learning-learning-to-achieve-goals-via-1","title":"C-Learning: Learning to Achieve Goals via Recursive Classification","date":"2020-11-17","arxiv_id":"2011.08909","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-model-free-reinforcement-learning","title":"Constrained Model-Free Reinforcement Learning for Process Optimization","date":"2020-11-16","arxiv_id":"2011.07925","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-q-learning-based-path-planning-and","title":"A deep Q-Learning based Path Planning and Navigation System for Firefighting Environments","date":"2020-11-12","arxiv_id":"2011.06450","repositories_listed":0,"syntology":null},{"url":null,"slug":"hamiltonian-q-learning-leveraging-importance-1","title":"On Using Hamiltonian Monte Carlo Sampling for Reinforcement Learning Problems in High-dimension","date":"2020-11-11","arxiv_id":"2011.05927","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-joint","title":"Multi-Agent Reinforcement Learning for Channel Assignment and Power Allocation in Platoon-Based C-V2X Systems","date":"2020-11-09","arxiv_id":"2011.04555","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-deep-markov-models-with","title":"Reinforced Deep Markov Models With Applications in Automatic Trading","date":"2020-11-09","arxiv_id":"2011.04391","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-assignment-problem","title":"Reinforcement Learning for Assignment problem","date":"2020-11-08","arxiv_id":"2011.03909","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hysteretic-q-learning-coordination","title":"A Hysteretic Q-learning Coordination Framework for Emerging Mobility Systems in Smart Cities","date":"2020-11-05","arxiv_id":"2011.03137","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepfoldit-a-deep-reinforcement-learning","title":"DeepFoldit -- A Deep Reinforcement Learning Neural Network Folding Proteins","date":"2020-10-28","arxiv_id":"2011.03442","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-analysis-of-decentralized","title":"Finite-Time Convergence Rates of Decentralized Stochastic Approximation with Applications in Multi-Agent and Multi-Task Learning","date":"2020-10-28","arxiv_id":"2010.15088","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-consumption-and-battery-aging","title":"Energy Consumption and Battery Aging Minimization Using a Q-learning Strategy for a Battery/Ultracapacitor Electric Vehicle","date":"2020-10-27","arxiv_id":"2010.14115","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-time-reduction-using-warm-start","title":"Learning Time Reduction Using Warm Start Methods for a Reinforcement Learning Based Supervisory Control in Hybrid Electric Vehicle Applications","date":"2020-10-27","arxiv_id":"2010.14575","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-and-service-priority-aware-trajectory","title":"Energy and Service-priority aware Trajectory Design for UAV-BSs using Double Q-Learning","date":"2020-10-26","arxiv_id":"2010.13346","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-reinforcement-learning-by-a-finite","title":"Enhancing reinforcement learning by a finite reward response filter with a case study in intelligent structural control","date":"2020-10-25","arxiv_id":"2010.15597","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-adiabatic-theorem-for-policy-tracking-with","title":"An Adiabatic Theorem for Policy Tracking with TD-learning","date":"2020-10-24","arxiv_id":"2010.12848","repositories_listed":0,"syntology":null},{"url":null,"slug":"stabilizing-transformer-based-action-sequence","title":"Stabilizing Transformer-Based Action Sequence Generation For Q-Learning","date":"2020-10-23","arxiv_id":"2010.12698","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-surrogate-q-learning-for-autonomous","title":"Deep Surrogate Q-Learning for Autonomous Driving","date":"2020-10-21","arxiv_id":"2010.11278","repositories_listed":0,"syntology":null}],"record_sha256":"2eb48d86f0588af10f08cd1f37d6bf5a0f5dacb52d6e035a3eb618a178d08d15","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}