{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/10","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":10,"pages_in_order":18,"rows_per_page":100,"rows":[901,1000],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning/papers/9","next":"/method/q-learning/papers/11","papers":[{"paper":null,"slug":"density-estimation-for-conservative-q","title":"Density Estimation for Conservative Q-Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"disentangling-generalization-in-reinforcement","title":"Disentangling Generalization in Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/explanation-aware-experience-replay-in-rule","slug":"explanation-aware-experience-replay-in-rule","title":"Explanation-Aware Experience Replay in Rule-Dense Environments","date":"2021-09-29","arxiv_id":"2109.14711","n_code_links":1,"syntology":null},{"paper":"/paper/hyperdqn-a-randomized-exploration-method-for","slug":"hyperdqn-a-randomized-exploration-method-for","title":"HyperDQN: A Randomized Exploration Method for Deep Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-explicit-credit-assignment-for-multi","title":"Learning Explicit Credit Assignment for Multi-agent Joint Q-learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/offline-reinforcement-learning-with-in-sample","slug":"offline-reinforcement-learning-with-in-sample","title":"Offline Reinforcement Learning with In-sample Q-Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/on-the-estimation-bias-in-double-q-learning-1","slug":"on-the-estimation-bias-in-double-q-learning-1","title":"On the Estimation Bias in Double Q-Learning","date":"2021-09-29","arxiv_id":"2109.14419","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["stilwell-git/doubly-bounded-q-learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"online-robust-reinforcement-learning-with","title":"Online Robust Reinforcement Learning with Model Uncertainty","date":"2021-09-29","arxiv_id":"2109.14523","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-learning-for-real-time-control-of","title":"Q-learning for real time control of heterogeneous microagent collectives","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"q-learning-scheduler-for-multi-task-learning","title":"Q-Learning Scheduler for Multi-Task Learning through the use of Histogram of Task Uncertainty","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-and-data-efficient-q-learning-by","title":"Robust and Data-efficient Q-learning by Composite Value-estimation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sbf-delta-2-exploration-for-reinforcement","title":"$\\sbf{\\delta^2}$-exploration for Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"text-generation-with-efficient-soft-q-1","title":"Text Generation with Efficient (Soft) $Q$-Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-guide-and-the-explorer-smart-agents-for","title":"The guide and the explorer: smart agents for resource-limited iterated batch reinforcement learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-unknown-aware-deep-q-learning","title":"Towards Unknown-aware Deep Q-Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unifying-top-down-and-bottom-up-for-recurrent","title":"Unifying Top-down and Bottom-up for Recurrent Visual Attention","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-with-adjustments","title":"Deep Reinforcement Learning with Adjustments","date":"2021-09-28","arxiv_id":"2109.13463","n_code_links":0,"syntology":null},{"paper":null,"slug":"smart-home-energy-management-sequence-to","title":"Smart Home Energy Management: Sequence-to-Sequence Load Forecasting and Q-Learning","date":"2021-09-25","arxiv_id":"2109.12440","n_code_links":0,"syntology":null},{"paper":"/paper/parameter-free-deterministic-reduction-of-the","slug":"parameter-free-deterministic-reduction-of-the","title":"Parameter-free Reduction of the Estimation Bias in Deep Reinforcement Learning for Deterministic Policy Gradients","date":"2021-09-24","arxiv_id":"2109.11788","n_code_links":1,"syntology":null},{"paper":null,"slug":"fetal-oxygen-delivery-and-consumption-and","title":"Fetal oxygen delivery and consumption and blood gases in relation to gestational age","date":"2021-09-23","arxiv_id":"2109.11616","n_code_links":0,"syntology":null},{"paper":"/paper/estimation-error-correction-in-deep","slug":"estimation-error-correction-in-deep","title":"Estimation Error Correction in Deep Reinforcement Learning for Deterministic Actor-Critic Methods","date":"2021-09-22","arxiv_id":"2109.10736","n_code_links":1,"syntology":null},{"paper":null,"slug":"mepg-a-minimalist-ensemble-policy-gradient","title":"MEPG: A Minimalist Ensemble Policy Gradient Framework for Deep Reinforcement Learning","date":"2021-09-22","arxiv_id":"2109.10552","n_code_links":0,"syntology":null},{"paper":null,"slug":"off-line-approximate-dynamic-programming-for","title":"Off-line approximate dynamic programming for the vehicle routing problem with a highly variable customer basis and stochastic demands","date":"2021-09-21","arxiv_id":"2109.10200","n_code_links":0,"syntology":null},{"paper":null,"slug":"greedy-unmixing-for-q-learning-in-multi-agent","title":"Greedy UnMixing for Q-Learning in Multi-Agent Reinforcement Learning","date":"2021-09-19","arxiv_id":"2109.09034","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-from-peers-transfer-reinforcement","title":"Learning from Peers: Deep Transfer Reinforcement Learning for Joint Radio and Cache Resource Allocation in 5G RAN Slicing","date":"2021-09-16","arxiv_id":"2109.07999","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-on-encrypted-data","title":"Reinforcement Learning on Encrypted Data","date":"2021-09-16","arxiv_id":"2109.08236","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-of-a-human-in-the-loop-policy","title":"Convergence of a Human-in-the-Loop Policy-Gradient Algorithm With Eligibility Trace Under Reward, Policy, and Advantage Feedback","date":"2021-09-15","arxiv_id":"2109.07054","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimal-cycling-of-a-heterogenous-battery","title":"Optimal Cycling of a Heterogenous Battery Bank via Reinforcement Learning","date":"2021-09-15","arxiv_id":"2109.07137","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-hierarchical-reinforcement-agents-for","title":"Deep hierarchical reinforcement agents for automated penetration testing","date":"2021-09-14","arxiv_id":"2109.06449","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-transformer-for-learning-driving","title":"Vision Transformer for Learning Driving Policies in Complex Multi-Agent Environments","date":"2021-09-14","arxiv_id":"2109.06514","n_code_links":0,"syntology":null},{"paper":"/paper/bootstrapped-meta-learning","slug":"bootstrapped-meta-learning","title":"Bootstrapped Meta-Learning","date":"2021-09-09","arxiv_id":"2109.04504","n_code_links":1,"syntology":null},{"paper":"/paper/memory-semantization-through-perturbed-and","slug":"memory-semantization-through-perturbed-and","title":"Learning cortical representations through perturbed and adversarial dreaming","date":"2021-09-09","arxiv_id":"2109.04261","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["NicoZenith/PAD"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"user-tampering-in-reinforcement-learning","title":"User Tampering in Reinforcement Learning Recommender Systems","date":"2021-09-09","arxiv_id":"2109.04083","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-simbad-active-landmark-based-self","title":"Deep SIMBAD: Active Landmark-based Self-localization Using Ranking -based Scene Descriptor","date":"2021-09-06","arxiv_id":"2109.02786","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-based-strategy-design-for-robot","title":"Learning-Based Strategy Design for Robot-Assisted Reminiscence Therapy Based on a Developed Model for People with Dementia","date":"2021-09-06","arxiv_id":"2109.02194","n_code_links":0,"syntology":null},{"paper":"/paper/temporal-aware-deep-reinforcement-learning","slug":"temporal-aware-deep-reinforcement-learning","title":"Temporal Shift Reinforcement Learning","date":"2021-09-05","arxiv_id":"2109.02145","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-communication-in-multi-agent-1","title":"Event-Based Communication in Distributed Q-Learning","date":"2021-09-03","arxiv_id":"2109.01417","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-at-the-edge-of","slug":"deep-reinforcement-learning-at-the-edge-of","title":"Deep Reinforcement Learning at the Edge of the Statistical Precipice","date":"2021-08-30","arxiv_id":"2108.13264","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 4 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/rliable"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-reinforcement-learning-for-dynamic-band","title":"Deep Reinforcement Learning for Dynamic Band Switch in Cellular-Connected UAV","date":"2021-08-26","arxiv_id":"2108.12054","n_code_links":0,"syntology":null},{"paper":null,"slug":"dqlel-deep-q-learning-for-energy-optimized","title":"DQLEL: Deep Q-Learning for Energy-Optimized LoS/NLoS UWB Node Selection","date":"2021-08-24","arxiv_id":"2108.13157","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-independent-study-of-reinforcement","title":"An Independent Study of Reinforcement Learning and Autonomous Driving","date":"2021-08-20","arxiv_id":"2110.07729","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-microscopic-pandemic-simulator-for-pandemic","title":"A Microscopic Pandemic Simulator for Pandemic Prediction Using Scalable Million-Agent Reinforcement Learning","date":"2021-08-14","arxiv_id":"2108.06589","n_code_links":0,"syntology":null},{"paper":"/paper/dqn-control-solution-for-kdd-cup-2021-city","slug":"dqn-control-solution-for-kdd-cup-2021-city","title":"DQN Control Solution for KDD Cup 2021 City Brain Challenge","date":"2021-08-14","arxiv_id":"2108.06491","n_code_links":1,"syntology":null},{"paper":null,"slug":"dq-gat-towards-safe-and-efficient-autonomous","title":"DQ-GAT: Towards Safe and Efficient Autonomous Driving with Deep Q-Learning and Graph Attention Networks","date":"2021-08-11","arxiv_id":"2108.05030","n_code_links":0,"syntology":null},{"paper":null,"slug":"two-is-a-crowd-tracking-relations-in-videos","title":"Two is a crowd: tracking relations in videos","date":"2021-08-11","arxiv_id":"2108.05331","n_code_links":0,"syntology":null},{"paper":null,"slug":"modified-double-dqn-addressing-stability","title":"Modified Double DQN: addressing stability","date":"2021-08-09","arxiv_id":"2108.04115","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-elementary-proof-that-q-learning-converges","title":"An Elementary Proof that Q-learning Converges Almost Surely","date":"2021-08-05","arxiv_id":"2108.02827","n_code_links":0,"syntology":null},{"paper":null,"slug":"offline-decentralized-multi-agent","title":"Offline Decentralized Multi-Agent Reinforcement Learning","date":"2021-08-04","arxiv_id":"2108.01832","n_code_links":0,"syntology":null},{"paper":"/paper/saber-data-driven-motion-planner-for","slug":"saber-data-driven-motion-planner-for","title":"SABER: Data-Driven Motion Planner for Autonomously Navigating Heterogeneous Robots","date":"2021-08-03","arxiv_id":"2108.01262","n_code_links":1,"syntology":null},{"paper":"/paper/an-efficient-image-to-image-translation","slug":"an-efficient-image-to-image-translation","title":"An Efficient Image-to-Image Translation HourGlass-based Architecture for Object Pushing Policy Learning","date":"2021-08-02","arxiv_id":"2108.01034","n_code_links":1,"syntology":null},{"paper":"/paper/a-dqn-based-approach-to-finding-precise","slug":"a-dqn-based-approach-to-finding-precise","title":"A DQN-based Approach to Finding Precise Evidences for Fact Verification","date":"2021-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"rem-efficient-semi-automated-real-time","title":"REM: Efficient Semi-Automated Real-Time Moderation of Online Forums","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"q-learning-for-conflict-resolution-in-b5g","title":"A Distributed Intelligence Architecture for B5G Network Automation","date":"2021-07-28","arxiv_id":"2107.13268","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-improved-algorithm-of-robot-path-planning","title":"An Improved Algorithm of Robot Path Planning in Complex Environment Based on Double DQN","date":"2021-07-23","arxiv_id":"2107.11245","n_code_links":0,"syntology":null},{"paper":null,"slug":"constraints-penalized-q-learning-for-safe","title":"Constraints Penalized Q-learning for Safe Offline Reinforcement Learning","date":"2021-07-19","arxiv_id":"2107.09003","n_code_links":0,"syntology":null},{"paper":"/paper/a-reinforcement-learning-environment-for-2","slug":"a-reinforcement-learning-environment-for-2","title":"A Reinforcement Learning Environment for Mathematical Reasoning via Program Synthesis","date":"2021-07-15","arxiv_id":"2107.07373","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-dynamic-2","title":"Deep Reinforcement Learning based Dynamic Optimization of Bus Timetable","date":"2021-07-15","arxiv_id":"2107.07066","n_code_links":0,"syntology":null},{"paper":null,"slug":"minimizing-safety-interference-for-safe-and","title":"Minimizing Safety Interference for Safe and Comfortable Automated Driving with Distributional Reinforcement Learning","date":"2021-07-15","arxiv_id":"2107.07316","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-penalized-shared-parameter-algorithm-for","title":"A Penalized Shared-parameter Algorithm for Estimating Optimal Dynamic Treatment Regimens","date":"2021-07-13","arxiv_id":"2107.07875","n_code_links":0,"syntology":null},{"paper":null,"slug":"transfer-learning-in-multi-agent","title":"Transfer Learning in Multi-Agent Reinforcement Learning with Double Q-Networks for Distributed Resource Sharing in V2X Communication","date":"2021-07-13","arxiv_id":"2107.06195","n_code_links":0,"syntology":null},{"paper":"/paper/backprop-free-reinforcement-learning-with","slug":"backprop-free-reinforcement-learning-with","title":"Backprop-Free Reinforcement Learning with Active Neural Generative Coding","date":"2021-07-10","arxiv_id":"2107.07046","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ago109/active-neural-generative-coding"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"reinforced-hybrid-genetic-algorithm-for-the","title":"Reinforced Hybrid Genetic Algorithm for the Traveling Salesman Problem","date":"2021-07-09","arxiv_id":"2107.06870","n_code_links":0,"syntology":null},{"paper":"/paper/computational-benefits-of-intermediate","slug":"computational-benefits-of-intermediate","title":"Computational Benefits of Intermediate Rewards for Goal-Reaching Policy Learning","date":"2021-07-08","arxiv_id":"2107.03961","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kebaek/minigrid"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ensemble-and-auxiliary-tasks-for-data","slug":"ensemble-and-auxiliary-tasks-for-data","title":"Ensemble and Auxiliary Tasks for Data-Efficient Deep Reinforcement Learning","date":"2021-07-05","arxiv_id":"2107.01904","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["NUS-LID/RENAULT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/improve-agents-without-retraining-parallel","slug":"improve-agents-without-retraining-parallel","title":"Improve Agents without Retraining: Parallel Tree Search with Off-Policy Correction","date":"2021-07-04","arxiv_id":"2107.01715","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-novel-deep-reinforcement-learning-based","title":"A Novel Deep Reinforcement Learning Based Stock Direction Prediction using Knowledge Graph and Community Aware Sentiments","date":"2021-07-02","arxiv_id":"2107.00931","n_code_links":0,"syntology":null},{"paper":null,"slug":"aoi-minimization-in-energy-harvesting-and","title":"AoI Minimization in Energy Harvesting and Spectrum Sharing Enabled 6G Networks","date":"2021-07-01","arxiv_id":"2107.00340","n_code_links":0,"syntology":null},{"paper":"/paper/distilling-reinforcement-learning-tricks-for","slug":"distilling-reinforcement-learning-tricks-for","title":"Distilling Reinforcement Learning Tricks for Video Games","date":"2021-07-01","arxiv_id":"2107.00703","n_code_links":1,"syntology":null},{"paper":null,"slug":"gap-dependent-bounds-for-two-player-markov","title":"Gap-Dependent Bounds for Two-Player Markov Games","date":"2021-07-01","arxiv_id":"2107.00685","n_code_links":0,"syntology":null},{"paper":null,"slug":"markov-decision-process-modeled-with-bandits","title":"Markov Decision Process modeled with Bandits for Sequential Decision Making in Linear-flow","date":"2021-07-01","arxiv_id":"2107.00204","n_code_links":0,"syntology":null},{"paper":"/paper/a-convergent-and-efficient-deep-q-network","slug":"a-convergent-and-efficient-deep-q-network","title":"Convergent and Efficient Deep Q Network Algorithm","date":"2021-06-29","arxiv_id":"2106.15419","n_code_links":1,"syntology":null},{"paper":null,"slug":"drill-deep-reinforcement-learning-for","title":"DRILL-- Deep Reinforcement Learning for Refinement Operators in $\\mathcal{ALC}$","date":"2021-06-29","arxiv_id":"2106.15373","n_code_links":0,"syntology":null},{"paper":null,"slug":"expert-q-learning-deep-q-learning-with-state","title":"Expert Q-learning: Deep Reinforcement Learning with Coarse State Values from Offline Expert Examples","date":"2021-06-28","arxiv_id":"2106.14642","n_code_links":0,"syntology":null},{"paper":null,"slug":"concentration-of-contractive-stochastic","title":"Concentration of Contractive Stochastic Approximation and Reinforcement Learning","date":"2021-06-27","arxiv_id":"2106.14308","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-mean-field-games","title":"Reinforcement Learning for Mean Field Games, with Applications to Economics","date":"2021-06-25","arxiv_id":"2106.13755","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploration-exploitation-in-multi-agent-1","title":"Exploration-Exploitation in Multi-Agent Competition: Convergence with Bounded Rationality","date":"2021-06-24","arxiv_id":"2106.12928","n_code_links":0,"syntology":null},{"paper":"/paper/coarse-to-fine-q-attention-efficient-learning","slug":"coarse-to-fine-q-attention-efficient-learning","title":"Coarse-to-Fine Q-attention: Efficient Learning for Visual Robotic Manipulation via Discretisation","date":"2021-06-23","arxiv_id":"2106.12534","n_code_links":1,"syntology":null},{"paper":null,"slug":"mmd-mix-value-function-factorisation-with","title":"MMD-MIX: Value Function Factorisation with Maximum Mean Discrepancy for Cooperative Multi-Agent Reinforcement Learning","date":"2021-06-22","arxiv_id":"2106.11652","n_code_links":0,"syntology":null},{"paper":"/paper/q-learning-lagrange-policies-for-multi-action","slug":"q-learning-lagrange-policies-for-multi-action","title":"Q-Learning Lagrange Policies for Multi-Action Restless Bandits","date":"2021-06-22","arxiv_id":"2106.12024","n_code_links":1,"syntology":{"ran":14,"of":17,"n_ran_checked":14,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["killian-34/MAIQL_and_LPQL"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/reinforcement-learning-for-phy-layer","slug":"reinforcement-learning-for-phy-layer","title":"Reinforcement Learning for Physical Layer Communications","date":"2021-06-22","arxiv_id":"2106.11595","n_code_links":1,"syntology":null},{"paper":null,"slug":"analytically-tractable-bayesian-deep-q","title":"Analytically Tractable Bayesian Deep Q-Learning","date":"2021-06-21","arxiv_id":"2106.11086","n_code_links":0,"syntology":null},{"paper":"/paper/distributed-heuristic-multi-agent-path","slug":"distributed-heuristic-multi-agent-path","title":"Distributed Heuristic Multi-Agent Path Finding with Communication","date":"2021-06-21","arxiv_id":"2106.11365","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-resource-1","title":"Reinforcement Learning for Resource Allocation in Steerable Laser-based Optical Wireless Systems","date":"2021-06-21","arxiv_id":"2106.11368","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deep-reinforcement-learning-approach","title":"A Deep Reinforcement Learning Approach towards Pendulum Swing-up Problem based on TF-Agents","date":"2021-06-17","arxiv_id":"2106.09556","n_code_links":0,"syntology":null},{"paper":null,"slug":"unbiased-methods-for-multi-goal-reinforcement","title":"Unbiased Methods for Multi-Goal Reinforcement Learning","date":"2021-06-16","arxiv_id":"2106.08863","n_code_links":0,"syntology":null},{"paper":"/paper/vision-language-navigation-with-random","slug":"vision-language-navigation-with-random","title":"Vision-Language Navigation with Random Environmental Mixup","date":"2021-06-15","arxiv_id":"2106.07876","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":0,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["lcfractal/vlnrem"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/text-generation-with-efficient-soft-q","slug":"text-generation-with-efficient-soft-q","title":"Efficient (Soft) Q-Learning for Text Generation with Limited Good Data","date":"2021-06-14","arxiv_id":"2106.07704","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["HanGuo97/soft-Q-learning-for-text-generation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/douzero-mastering-doudizhu-with-self-play","slug":"douzero-mastering-doudizhu-with-self-play","title":"DouZero: Mastering DouDizhu with Self-Play Deep Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06135","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":2,"n_instrument":1,"unverified":3,"pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["kwai/DouZero"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"paper":"/paper/gdi-rethinking-what-makes-reinforcement","slug":"gdi-rethinking-what-makes-reinforcement","title":"GDI: Rethinking What Makes Reinforcement Learning Different From Supervised Learning","date":"2021-06-11","arxiv_id":"2106.06232","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforced-few-shot-acquisition-function","title":"Reinforced Few-Shot Acquisition Function Learning for Bayesian Optimization","date":"2021-06-08","arxiv_id":"2106.04335","n_code_links":0,"syntology":null},{"paper":"/paper/believe-what-you-see-implicit-constraint","slug":"believe-what-you-see-implicit-constraint","title":"Believe What You See: Implicit Constraint Approach for Offline Multi-Agent Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03400","n_code_links":1,"syntology":null},{"paper":null,"slug":"decentralized-q-learning-in-zero-sum-markov","title":"Decentralized Q-Learning in Zero-sum Markov Games","date":"2021-06-04","arxiv_id":"2106.02748","n_code_links":0,"syntology":null},{"paper":null,"slug":"design-and-comparison-of-reward-functions-in","title":"Design and Comparison of Reward Functions in Reinforcement Learning for Energy Management of Sensor Nodes","date":"2021-06-02","arxiv_id":"2106.01114","n_code_links":0,"syntology":null},{"paper":null,"slug":"smooth-q-learning-accelerate-convergence-of-q","title":"Smooth Q-learning: Accelerate Convergence of Q-learning Using Similarity","date":"2021-06-02","arxiv_id":"2106.01134","n_code_links":0,"syntology":null},{"paper":null,"slug":"energy-aware-placement-optimization-of-uav","title":"Energy-aware optimization of UAV base stations placement via decentralized multi-agent Q-learning","date":"2021-06-01","arxiv_id":"2106.00845","n_code_links":0,"syntology":null},{"paper":"/paper/shaq-incorporating-shapley-value-theory-into","slug":"shaq-incorporating-shapley-value-theory-into","title":"SHAQ: Incorporating Shapley Value Theory into Multi-Agent Q-Learning","date":"2021-05-31","arxiv_id":"2105.15013","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["hsvgbkhgbv/shapley-q-learning"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"sample-efficient-reinforcement-learning-for","title":"Sample-Efficient Reinforcement Learning for Linearly-Parameterized MDPs with a Generative Model","date":"2021-05-28","arxiv_id":"2105.14016","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-optimize-industry-scale-dynamic","title":"Learning to Optimize Industry-Scale Dynamic Pickup and Delivery Problems","date":"2021-05-27","arxiv_id":"2105.12899","n_code_links":0,"syntology":null},{"paper":null,"slug":"reputation-bootstrapping-for-composite","title":"Reputation Bootstrapping for Composite Services using CP-nets","date":"2021-05-27","arxiv_id":"2105.15135","n_code_links":0,"syntology":null},{"paper":null,"slug":"verification-of-dissipativity-and-evaluation","title":"Verification of Dissipativity and Evaluation of Storage Function in Economic Nonlinear MPC using Q-Learning","date":"2021-05-24","arxiv_id":"2105.11313","n_code_links":0,"syntology":null}],"record_sha256":"0232360f98f03ed507f6e9858b80cd2df9be163533379ad50d5bb1b8c62154cd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}