{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/q-learning/papers/13","list_of":"/task/q-learning","task":"Q-Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":13,"pages_in_order":20,"rows_per_page":100,"rows":[1201,1300],"of":1918,"counts":{"archive_papers_tagged":1918,"with_a_code_link":463,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1918,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":102,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":102,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/q-learning","prev":"/task/q-learning/papers/12","next":"/task/q-learning/papers/14","papers":[{"url":null,"slug":"fast-block-linear-system-solver-using-q","title":"Fast Block Linear System Solver Using Q-Learning Schduling for Unified Dynamic Power System Simulations","date":"2021-10-12","arxiv_id":"2110.05843","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-reinforcement-learning-in","title":"Provably Efficient Reinforcement Learning in Decentralized General-Sum Markov Games","date":"2021-10-12","arxiv_id":"2110.05682","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigation-in-urban-environments-amongst","title":"Navigation In Urban Environments Amongst Pedestrians Using Multi-Objective Deep Reinforcement Learning","date":"2021-10-11","arxiv_id":"2110.05205","repositories_listed":0,"syntology":null},{"url":null,"slug":"urban-traffic-dynamic-rerouting-framework-a","title":"Urban traffic dynamic rerouting framework: A DRL-based model with fog-cloud architecture","date":"2021-10-11","arxiv_id":"2110.05532","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-learning-inference-scheme-based-on","title":"A Deep Learning Inference Scheme Based on Pipelined Matrix Multiplication Acceleration Design and Non-uniform Quantization","date":"2021-10-10","arxiv_id":"2110.04861","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-the-sample-complexity-barrier-to","title":"Breaking the Sample Complexity Barrier to Regret-Optimal Model-Free Reinforcement Learning","date":"2021-10-09","arxiv_id":"2110.04645","repositories_listed":0,"syntology":null},{"url":null,"slug":"nested-policy-reinforcement-learning","title":"Compositional Q-learning for electrolyte repletion with imbalanced patient sub-populations","date":"2021-10-06","arxiv_id":"2110.02879","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-framework-for-4","title":"A Deep Reinforcement Learning Framework for Contention-Based Spectrum Sharing","date":"2021-10-05","arxiv_id":"2110.02736","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-first-passage-time-minimization","title":"A study of first-passage time minimization via Q-learning in heated gridworlds","date":"2021-10-05","arxiv_id":"2110.02129","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-guidewire","title":"Deep reinforcement learning for guidewire navigation in coronary artery phantom","date":"2021-10-05","arxiv_id":"2110.01840","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-modified-q-learning-algorithm-for-rate","title":"A Modified Q-Learning Algorithm for Rate-Profiling of Polarization Adjusted Convolutional (PAC) Codes","date":"2021-10-04","arxiv_id":"2110.01563","repositories_listed":0,"syntology":null},{"url":null,"slug":"cellular-traffic-offloading-via-opportunistic","title":"Cellular traffic offloading via Opportunistic Networking with Reinforcement Learning","date":"2021-10-01","arxiv_id":"2110.00397","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-q-learning-for-interaction-limited","title":"Adaptive Q-learning for Interaction-Limited Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-attempt-to-model-human-trust-with","title":"An Attempt to Model Human Trust with Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapped-hindsight-experience-replay-with","title":"Bootstrapped Hindsight Experience replay with Counterintuitive Prioritization","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-deep-q-learning-in-optimal-control","title":"Continuous Deep Q-Learning in Optimal Control Problems: Normalized Advantage Functions Analysis","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"convergent-and-efficient-deep-q-learning","title":"Convergent and Efficient Deep Q Learning Algorithm","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-cooperative-multi-agent","title":"Decentralized Cooperative Multi-Agent Reinforcement Learning with Exploration","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"density-estimation-for-conservative-q","title":"Density Estimation for Conservative Q-Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-explicit-credit-assignment-for-multi","title":"Learning Explicit Credit Assignment for Multi-agent Joint Q-learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"online-robust-reinforcement-learning-with","title":"Online Robust Reinforcement Learning with Model Uncertainty","date":"2021-09-29","arxiv_id":"2109.14523","repositories_listed":0,"syntology":null},{"url":null,"slug":"polyphonic-music-composition-an-adversarial","title":"Polyphonic Music Composition: An Adversarial Inverse Reinforcement Learning Approach","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-for-real-time-control-of","title":"Q-learning for real time control of heterogeneous microagent collectives","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-scheduler-for-multi-task-learning","title":"Q-Learning Scheduler for Multi-Task Learning through the use of Histogram of Task Uncertainty","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-and-data-efficient-q-learning-by","title":"Robust and Data-efficient Q-learning by Composite Value-estimation","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sbf-delta-2-exploration-for-reinforcement","title":"$\\sbf{\\delta^2}$-exploration for Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"text-generation-with-efficient-soft-q-1","title":"Text Generation with Efficient (Soft) $Q$-Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-unknown-aware-deep-q-learning","title":"Towards Unknown-aware Deep Q-Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unifying-top-down-and-bottom-up-for-recurrent","title":"Unifying Top-down and Bottom-up for Recurrent Visual Attention","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"untangling-braids-with-multi-agent-q-learning","title":"Untangling Braids with Multi-agent Q-Learning","date":"2021-09-29","arxiv_id":"2109.14502","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-refinement-network-vrn","title":"Value Refinement Network (VRN)","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-adjustments","title":"Deep Reinforcement Learning with Adjustments","date":"2021-09-28","arxiv_id":"2109.13463","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-home-energy-management-sequence-to","title":"Smart Home Energy Management: Sequence-to-Sequence Load Forecasting and Q-Learning","date":"2021-09-25","arxiv_id":"2109.12440","repositories_listed":0,"syntology":null},{"url":null,"slug":"mepg-a-minimalist-ensemble-policy-gradient","title":"MEPG: A Minimalist Ensemble Policy Gradient Framework for Deep Reinforcement Learning","date":"2021-09-22","arxiv_id":"2109.10552","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-line-approximate-dynamic-programming-for","title":"Off-line approximate dynamic programming for the vehicle routing problem with a highly variable customer basis and stochastic demands","date":"2021-09-21","arxiv_id":"2109.10200","repositories_listed":0,"syntology":null},{"url":null,"slug":"search-for-deep-graph-neural-networks","title":"Search For Deep Graph Neural Networks","date":"2021-09-21","arxiv_id":"2109.10047","repositories_listed":0,"syntology":null},{"url":null,"slug":"greedy-unmixing-for-q-learning-in-multi-agent","title":"Greedy UnMixing for Q-Learning in Multi-Agent Reinforcement Learning","date":"2021-09-19","arxiv_id":"2109.09034","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularize-don-t-mix-multi-agent","title":"Regularize! Don't Mix: Multi-Agent Reinforcement Learning without Explicit Centralized Structures","date":"2021-09-19","arxiv_id":"2109.09038","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-peers-transfer-reinforcement","title":"Learning from Peers: Deep Transfer Reinforcement Learning for Joint Radio and Cache Resource Allocation in 5G RAN Slicing","date":"2021-09-16","arxiv_id":"2109.07999","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-of-a-human-in-the-loop-policy","title":"Convergence of a Human-in-the-Loop Policy-Gradient Algorithm With Eligibility Trace Under Reward, Policy, and Advantage Feedback","date":"2021-09-15","arxiv_id":"2109.07054","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-cycling-of-a-heterogenous-battery","title":"Optimal Cycling of a Heterogenous Battery Bank via Reinforcement Learning","date":"2021-09-15","arxiv_id":"2109.07137","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-hierarchical-reinforcement-agents-for","title":"Deep hierarchical reinforcement agents for automated penetration testing","date":"2021-09-14","arxiv_id":"2109.06449","repositories_listed":0,"syntology":null},{"url":null,"slug":"user-tampering-in-reinforcement-learning","title":"User Tampering in Reinforcement Learning Recommender Systems","date":"2021-09-09","arxiv_id":"2109.04083","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-of-batch-asynchronous-stochastic","title":"Convergence of Batch Asynchronous Stochastic Approximation With Applications to Reinforcement Learning","date":"2021-09-08","arxiv_id":"2109.03445","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-simbad-active-landmark-based-self","title":"Deep SIMBAD: Active Landmark-based Self-localization Using Ranking -based Scene Descriptor","date":"2021-09-06","arxiv_id":"2109.02786","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-strategy-design-for-robot","title":"Learning-Based Strategy Design for Robot-Assisted Reminiscence Therapy Based on a Developed Model for People with Dementia","date":"2021-09-06","arxiv_id":"2109.02194","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-communication-in-multi-agent-1","title":"Event-Based Communication in Distributed Q-Learning","date":"2021-09-03","arxiv_id":"2109.01417","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-dynamic-band","title":"Deep Reinforcement Learning for Dynamic Band Switch in Cellular-Connected UAV","date":"2021-08-26","arxiv_id":"2108.12054","repositories_listed":0,"syntology":null},{"url":null,"slug":"dqlel-deep-q-learning-for-energy-optimized","title":"DQLEL: Deep Q-Learning for Energy-Optimized LoS/NLoS UWB Node Selection","date":"2021-08-24","arxiv_id":"2108.13157","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-independent-study-of-reinforcement","title":"An Independent Study of Reinforcement Learning and Autonomous Driving","date":"2021-08-20","arxiv_id":"2110.07729","repositories_listed":0,"syntology":null},{"url":null,"slug":"dq-gat-towards-safe-and-efficient-autonomous","title":"DQ-GAT: Towards Safe and Efficient Autonomous Driving with Deep Q-Learning and Graph Attention Networks","date":"2021-08-11","arxiv_id":"2108.05030","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-maximize-influence","title":"Maximizing Influence with Graph Neural Networks","date":"2021-08-10","arxiv_id":"2108.04623","repositories_listed":0,"syntology":null},{"url":null,"slug":"modified-double-dqn-addressing-stability","title":"Modified Double DQN: addressing stability","date":"2021-08-09","arxiv_id":"2108.04115","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-elementary-proof-that-q-learning-converges","title":"An Elementary Proof that Q-learning Converges Almost Surely","date":"2021-08-05","arxiv_id":"2108.02827","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-decentralized-multi-agent","title":"Offline Decentralized Multi-Agent Reinforcement Learning","date":"2021-08-04","arxiv_id":"2108.01832","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-for-conflict-resolution-in-b5g","title":"A Distributed Intelligence Architecture for B5G Network Automation","date":"2021-07-28","arxiv_id":"2107.13268","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-based-reinforcement-learning-for","title":"Value-Based Reinforcement Learning for Continuous Control Robotic Manipulation in Multi-Task Sparse Reward Settings","date":"2021-07-28","arxiv_id":"2107.13356","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-deep-q-learning-based-real-time","title":"Double Deep Q-learning Based Real-Time Optimization Strategy for Microgrids","date":"2021-07-27","arxiv_id":"2107.12545","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-deep-learning-and-augmented","title":"Integrating Deep Learning and Augmented Reality to Enhance Situational Awareness in Firefighting Environments","date":"2021-07-23","arxiv_id":"2107.11043","repositories_listed":0,"syntology":null},{"url":null,"slug":"constraints-penalized-q-learning-for-safe","title":"Constraints Penalized Q-learning for Safe Offline Reinforcement Learning","date":"2021-07-19","arxiv_id":"2107.09003","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-penalized-shared-parameter-algorithm-for","title":"A Penalized Shared-parameter Algorithm for Estimating Optimal Dynamic Treatment Regimens","date":"2021-07-13","arxiv_id":"2107.07875","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-smash-q-learning-based-self-adaptation-of","title":"Q-SMASH: Q-Learning-based Self-Adaptation of Human-Centered Internet of Things","date":"2021-07-13","arxiv_id":"2107.05949","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-in-multi-agent","title":"Transfer Learning in Multi-Agent Reinforcement Learning with Double Q-Networks for Distributed Resource Sharing in V2X Communication","date":"2021-07-13","arxiv_id":"2107.06195","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-hybrid-genetic-algorithm-for-the","title":"Reinforced Hybrid Genetic Algorithm for the Traveling Salesman Problem","date":"2021-07-09","arxiv_id":"2107.06870","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-least-restriction-for-offline","title":"The Least Restriction for Offline Reinforcement Learning","date":"2021-07-05","arxiv_id":"2107.01757","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-deep-reinforcement-learning-based","title":"A Novel Deep Reinforcement Learning Based Stock Direction Prediction using Knowledge Graph and Community Aware Sentiments","date":"2021-07-02","arxiv_id":"2107.00931","repositories_listed":0,"syntology":null},{"url":null,"slug":"gap-dependent-bounds-for-two-player-markov","title":"Gap-Dependent Bounds for Two-Player Markov Games","date":"2021-07-01","arxiv_id":"2107.00685","repositories_listed":0,"syntology":null},{"url":null,"slug":"markov-decision-process-modeled-with-bandits","title":"Markov Decision Process modeled with Bandits for Sequential Decision Making in Linear-flow","date":"2021-07-01","arxiv_id":"2107.00204","repositories_listed":0,"syntology":null},{"url":null,"slug":"drill-deep-reinforcement-learning-for","title":"DRILL-- Deep Reinforcement Learning for Refinement Operators in $\\mathcal{ALC}$","date":"2021-06-29","arxiv_id":"2106.15373","repositories_listed":0,"syntology":null},{"url":null,"slug":"expert-q-learning-deep-q-learning-with-state","title":"Expert Q-learning: Deep Reinforcement Learning with Coarse State Values from Offline Expert Examples","date":"2021-06-28","arxiv_id":"2106.14642","repositories_listed":0,"syntology":null},{"url":null,"slug":"instance-optimality-in-optimal-value","title":"Instance-optimality in optimal value estimation: Adaptivity via variance-reduced Q-learning","date":"2021-06-28","arxiv_id":"2106.14352","repositories_listed":0,"syntology":null},{"url":null,"slug":"concentration-of-contractive-stochastic","title":"Concentration of Contractive Stochastic Approximation and Reinforcement Learning","date":"2021-06-27","arxiv_id":"2106.14308","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-mean-field-games","title":"Reinforcement Learning for Mean Field Games, with Applications to Economics","date":"2021-06-25","arxiv_id":"2106.13755","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-exploitation-in-multi-agent-1","title":"Exploration-Exploitation in Multi-Agent Competition: Convergence with Bounded Rationality","date":"2021-06-24","arxiv_id":"2106.12928","repositories_listed":0,"syntology":null},{"url":null,"slug":"analytically-tractable-bayesian-deep-q","title":"Analytically Tractable Bayesian Deep Q-Learning","date":"2021-06-21","arxiv_id":"2106.11086","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-resource-1","title":"Reinforcement Learning for Resource Allocation in Steerable Laser-based Optical Wireless Systems","date":"2021-06-21","arxiv_id":"2106.11368","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-offline-reinforcement-learning-with","title":"Boosting Offline Reinforcement Learning with Residual Generative Modeling","date":"2021-06-19","arxiv_id":"2106.10411","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach","title":"A Deep Reinforcement Learning Approach towards Pendulum Swing-up Problem based on TF-Agents","date":"2021-06-17","arxiv_id":"2106.09556","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-automated","title":"Deep reinforcement learning with automated label extraction from clinical reports accurately classifies 3D MRI brain volumes","date":"2021-06-17","arxiv_id":"2106.09812","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-q-learning-based-topology-aware-routing","title":"A Q-Learning-Based Topology-Aware Routing Protocol for Flying Ad Hoc Networks","date":"2021-06-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unbiased-methods-for-multi-goal-reinforcement","title":"Unbiased Methods for Multi-Goal Reinforcement Learning","date":"2021-06-16","arxiv_id":"2106.08863","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-q-learning-in-zero-sum-markov","title":"Decentralized Q-Learning in Zero-sum Markov Games","date":"2021-06-04","arxiv_id":"2106.02748","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-and-comparison-of-reward-functions-in","title":"Design and Comparison of Reward Functions in Reinforcement Learning for Energy Management of Sensor Nodes","date":"2021-06-02","arxiv_id":"2106.01114","repositories_listed":0,"syntology":null},{"url":null,"slug":"smooth-q-learning-accelerate-convergence-of-q","title":"Smooth Q-learning: Accelerate Convergence of Q-learning Using Similarity","date":"2021-06-02","arxiv_id":"2106.01134","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-to-improve","title":"A reinforcement learning approach to improve communication performance and energy utilization in fog-based IoT","date":"2021-06-01","arxiv_id":"2106.00654","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-aware-placement-optimization-of-uav","title":"Energy-aware optimization of UAV base stations placement via decentralized multi-agent Q-learning","date":"2021-06-01","arxiv_id":"2106.00845","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-for","title":"Sample-Efficient Reinforcement Learning for Linearly-Parameterized MDPs with a Generative Model","date":"2021-05-28","arxiv_id":"2105.14016","repositories_listed":0,"syntology":null},{"url":null,"slug":"reputation-bootstrapping-for-composite","title":"Reputation Bootstrapping for Composite Services using CP-nets","date":"2021-05-27","arxiv_id":"2105.15135","repositories_listed":0,"syntology":null},{"url":null,"slug":"verification-of-dissipativity-and-evaluation","title":"Verification of Dissipativity and Evaluation of Storage Function in Economic Nonlinear MPC using Q-Learning","date":"2021-05-24","arxiv_id":"2105.11313","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-adaptive-optimal-control-algorithm","title":"Online Adaptive Optimal Control Algorithm Based on Synchronous Integral Reinforcement Learning With Explorations","date":"2021-05-19","arxiv_id":"2105.09006","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparsity-prior-regularized-q-learning-for","title":"Reinforcement Learning With Sparse-Executing Actions via Sparsity Regularization","date":"2021-05-18","arxiv_id":"2105.08666","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-off-policy-q-learning-for-data","title":"Efficient Off-Policy Q-Learning for Data-Based Discrete-Time LQR Problems","date":"2021-05-17","arxiv_id":"2105.07761","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-intervene-an-adaptive-learning","title":"Learn to Intervene: An Adaptive Learning Policy for Restless Bandits in Application to Preventive Healthcare","date":"2021-05-17","arxiv_id":"2105.07965","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-performance-analysis-towards","title":"Interpretable performance analysis towards offline reinforcement learning: A dataset perspective","date":"2021-05-12","arxiv_id":"2105.05473","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-constraint-satisfaction-problem-and","title":"Fast constraint satisfaction problem and learning-based algorithm for solving Minesweeper","date":"2021-05-10","arxiv_id":"2105.04120","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-expert-trajectory","title":"Reinforcement Learning with Expert Trajectory For Quantitative Trading","date":"2021-05-09","arxiv_id":"2105.03844","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-on-multi-agent-q-learning-frameworks","title":"Survey on Multi-Agent Q-Learning frameworks for resource management in wireless sensor network","date":"2021-05-05","arxiv_id":"2105.02371","repositories_listed":0,"syntology":null},{"url":null,"slug":"carl-dtn-context-adaptive-reinforcement","title":"CARL-DTN: Context Adaptive Reinforcement Learning based Routing Algorithm in Delay Tolerant Network","date":"2021-05-02","arxiv_id":"2105.00544","repositories_listed":0,"syntology":null},{"url":null,"slug":"rp-dqn-an-application-of-q-learning-to","title":"RP-DQN: An application of Q-Learning to Vehicle Routing Problems","date":"2021-04-25","arxiv_id":"2104.12226","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-aided-deep-reinforcement-learning-for","title":"Model-aided Deep Reinforcement Learning for Sample-efficient UAV Trajectory Design in IoT Networks","date":"2021-04-21","arxiv_id":"2104.10403","repositories_listed":0,"syntology":null}],"record_sha256":"85635d79307b788b3b77c672ec5eb8177c69a34017af7f967c55fc3375c43caf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}