{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/13","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":13,"pages_in_order":18,"rows_per_page":100,"rows":[1201,1300],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning/papers/12","next":"/method/q-learning/papers/14","papers":[{"paper":"/paper/reward-machines-for-cooperative-multi-agent","slug":"reward-machines-for-cooperative-multi-agent","title":"Reward Machines for Cooperative Multi-Agent Reinforcement Learning","date":"2020-07-03","arxiv_id":"2007.01962","n_code_links":2,"syntology":null},{"paper":null,"slug":"decentralized-deep-reinforcement-learning-for-1","title":"Decentralized Deep Reinforcement Learning for Network Level Traffic Signal Control","date":"2020-07-02","arxiv_id":"2007.03433","n_code_links":0,"syntology":null},{"paper":"/paper/gradient-temporal-difference-learning-with","slug":"gradient-temporal-difference-learning-with","title":"Gradient Temporal-Difference Learning with Regularized Corrections","date":"2020-07-01","arxiv_id":"2007.00611","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rlai-lab/Regularized-GradientTD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/group-equivariant-deep-reinforcement-learning","slug":"group-equivariant-deep-reinforcement-learning","title":"Group Equivariant Deep Reinforcement Learning","date":"2020-07-01","arxiv_id":"2007.03437","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["arnab39/EquivariantDQN"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"regularly-updated-deterministic-policy","title":"Regularly Updated Deterministic Policy Gradient Algorithm","date":"2020-07-01","arxiv_id":"2007.00169","n_code_links":0,"syntology":null},{"paper":null,"slug":"provably-more-efficient-q-learning-in-the","title":"Provably More Efficient Q-Learning in the One-Sided-Feedback/Full-Feedback Settings","date":"2020-06-30","arxiv_id":"2007.00080","n_code_links":0,"syntology":null},{"paper":null,"slug":"concept-and-the-implementation-of-a-tool-to","title":"Concept and the implementation of a tool to convert industry 4.0 environments modeled as FSM to an OpenAI Gym wrapper","date":"2020-06-29","arxiv_id":"2006.16035","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-reinforcement-learning-to-herd-a","title":"Using Reinforcement Learning to Herd a Robotic Swarm to a Target Distribution","date":"2020-06-29","arxiv_id":"2006.15807","n_code_links":0,"syntology":null},{"paper":"/paper/image-classification-by-reinforcement","slug":"image-classification-by-reinforcement","title":"Image Classification by Reinforcement Learning with Two-State Q-Learning","date":"2020-06-28","arxiv_id":"2007.01298","n_code_links":1,"syntology":null},{"paper":"/paper/lookahead-bounded-q-learning","slug":"lookahead-bounded-q-learning","title":"Lookahead-Bounded Q-Learning","date":"2020-06-28","arxiv_id":"2006.15690","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-handwritten","title":"Reinforcement Learning Based Handwritten Digit Recognition with Two-State Q-Learning","date":"2020-06-28","arxiv_id":"2007.01193","n_code_links":0,"syntology":null},{"paper":"/paper/overfitting-and-optimization-in-offline","slug":"overfitting-and-optimization-in-offline","title":"Offline Contextual Bandits with Overparameterized Models","date":"2020-06-27","arxiv_id":"2006.15368","n_code_links":1,"syntology":null},{"paper":null,"slug":"q-learning-with-differential-entropy-of-q","title":"Q-Learning with Differential Entropy of Q-Tables","date":"2020-06-26","arxiv_id":"2006.14795","n_code_links":0,"syntology":null},{"paper":null,"slug":"noise-overestimation-and-exploration-in-deep","title":"Some approaches used to overcome overestimation in Deep Reinforcement Learning algorithms","date":"2020-06-25","arxiv_id":"2006.14167","n_code_links":0,"syntology":null},{"paper":null,"slug":"reducing-overestimation-bias-by-increasing","title":"Preventing Value Function Collapse in Ensemble {Q}-Learning by Maximizing Representation Diversity","date":"2020-06-24","arxiv_id":"2006.13823","n_code_links":0,"syntology":null},{"paper":"/paper/rl-unplugged-benchmarks-for-offline","slug":"rl-unplugged-benchmarks-for-offline","title":"RL Unplugged: A Suite of Benchmarks for Offline Reinforcement Learning","date":"2020-06-24","arxiv_id":"2006.13888","n_code_links":2,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-control-for-radar","title":"Deep Reinforcement Learning Control for Radar Detection and Tracking in Congested Spectral Environments","date":"2020-06-23","arxiv_id":"2006.13173","n_code_links":0,"syntology":null},{"paper":null,"slug":"risk-sensitive-reinforcement-learning-near","title":"Risk-Sensitive Reinforcement Learning: Near-Optimal Risk-Sample Tradeoff in Regret","date":"2020-06-22","arxiv_id":"2006.13827","n_code_links":0,"syntology":null},{"paper":null,"slug":"nrowan-dqn-a-stable-noisy-network-with-noise","title":"NROWAN-DQN: A Stable Noisy Network with Noise Reduction and Online Weight Adjustment for Exploration","date":"2020-06-19","arxiv_id":"2006.10980","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-ridesharing-dispatch-using-multi","slug":"efficient-ridesharing-dispatch-using-multi","title":"Efficient Ridesharing Dispatch Using Multi-Agent Reinforcement Learning","date":"2020-06-18","arxiv_id":"2006.10897","n_code_links":1,"syntology":null},{"paper":null,"slug":"parameterized-mdps-and-reinforcement-learning","title":"Parameterized MDPs and Reinforcement Learning Problems -- A Maximum Entropy Principle Based Framework","date":"2020-06-17","arxiv_id":"2006.09646","n_code_links":0,"syntology":null},{"paper":"/paper/semantic-visual-navigation-by-watching","slug":"semantic-visual-navigation-by-watching","title":"Semantic Visual Navigation by Watching YouTube Videos","date":"2020-06-17","arxiv_id":"2006.10034","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-teaching-dimension-of-q-learning","title":"The Sample Complexity of Teaching-by-Reinforcement on Q-Learning","date":"2020-06-16","arxiv_id":"2006.09324","n_code_links":0,"syntology":null},{"paper":null,"slug":"interaction-networks-using-a-reinforcement","title":"Interaction Networks: Using a Reinforcement Learner to train other Machine Learning algorithms","date":"2020-06-15","arxiv_id":"2006.08457","n_code_links":0,"syntology":null},{"paper":null,"slug":"decorrelated-double-q-learning","title":"Decorrelated Double Q-learning","date":"2020-06-12","arxiv_id":"2006.06956","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-neural","title":"Deep Reinforcement Learning for Neural Control","date":"2020-06-12","arxiv_id":"2006.07352","n_code_links":0,"syntology":null},{"paper":null,"slug":"human-and-multi-agent-collaboration-in-a","title":"Human and Multi-Agent collaboration in a human-MARL teaming framework","date":"2020-06-12","arxiv_id":"2006.07301","n_code_links":0,"syntology":null},{"paper":null,"slug":"safety-guaranteed-reinforcement-learning","title":"Safety-guaranteed Reinforcement Learning based on Multi-class Support Vector Machine","date":"2020-06-12","arxiv_id":"2006.07446","n_code_links":0,"syntology":null},{"paper":"/paper/self-imitation-learning-via-generalized-lower","slug":"self-imitation-learning-via-generalized-lower","title":"Self-Imitation Learning via Generalized Lower Bound Q-learning","date":"2020-06-12","arxiv_id":"2006.07442","n_code_links":0,"syntology":{"ran":16,"of":20,"n_ran_checked":13,"n_instrument":3,"unverified":4,"pointer_only":11,"phrase":"16 ran (of which 4 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 2 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":null}},{"paper":null,"slug":"exploration-by-maximizing-renyi-entropy-for","title":"Exploration by Maximizing Rényi Entropy for Reward-Free RL Framework","date":"2020-06-11","arxiv_id":"2006.06193","n_code_links":0,"syntology":null},{"paper":null,"slug":"fitted-q-learning-for-relational-domains","title":"Fitted Q-Learning for Relational Domains","date":"2020-06-10","arxiv_id":"2006.05595","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-free-algorithm-and-regret-analysis-for-1","title":"Model-Free Algorithm and Regret Analysis for MDPs with Long-Term Constraints","date":"2020-06-10","arxiv_id":"2006.05961","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-agent-reinforcement-learning-in-a","title":"Multi-Agent Reinforcement Learning in a Realistic Limit Order Book Market Simulation","date":"2020-06-10","arxiv_id":"2006.05574","n_code_links":0,"syntology":null},{"paper":null,"slug":"privacy-cost-management-in-smart-meters-an","title":"Privacy-Cost Management in Smart Meters with Mutual Information-Based Reinforcement Learning","date":"2020-06-10","arxiv_id":"2006.06106","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-joint-self","title":"Reinforcement Learning-Based Joint Self-Optimisation Method for the Fuzzy Logic Handover Algorithm in 5G HetNets","date":"2020-06-09","arxiv_id":"2006.05010","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-model-free-learning-algorithm-for-infinite","title":"A Model-free Learning Algorithm for Infinite-horizon Average-reward MDPs with Near-optimal Regret","date":"2020-06-08","arxiv_id":"2006.04354","n_code_links":0,"syntology":null},{"paper":null,"slug":"balancing-a-cartpole-system-with","title":"Balancing a CartPole System with Reinforcement Learning -- A Tutorial","date":"2020-06-08","arxiv_id":"2006.04938","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-temporal-difference-and-q-learning-learn","title":"Can Temporal-Difference and Q-Learning Learn Representation? A Mean-Field Theory","date":"2020-06-08","arxiv_id":"2006.04761","n_code_links":0,"syntology":null},{"paper":"/paper/conservative-q-learning-for-offline","slug":"conservative-q-learning-for-offline","title":"Conservative Q-Learning for Offline Reinforcement Learning","date":"2020-06-08","arxiv_id":"2006.04779","n_code_links":18,"syntology":{"ran":31,"of":34,"n_ran_checked":28,"n_instrument":3,"unverified":3,"pointer_only":19,"phrase":"31 ran (of which 5 constructed an object rather than computing a result; 28 with no instrument failure: 2 honoured, 0 violated, 26 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["aviralkumar2907/CQL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"a-multi-step-and-resilient-predictive-q","title":"A Multi-step and Resilient Predictive Q-learning Algorithm for IoT with Human Operators in the Loop: A Case Study in Water Supply Networks","date":"2020-06-06","arxiv_id":"2006.03899","n_code_links":0,"syntology":null},{"paper":null,"slug":"logical-team-q-learning-an-approach-towards","title":"Logical Team Q-learning: An approach towards factored policies in cooperative MARL","date":"2020-06-05","arxiv_id":"2006.03553","n_code_links":0,"syntology":null},{"paper":"/paper/a-novel-update-mechanism-for-q-networks-based","slug":"a-novel-update-mechanism-for-q-networks-based","title":"A Novel Update Mechanism for Q-Networks Based On Extreme Learning Machines","date":"2020-06-04","arxiv_id":"2006.02986","n_code_links":1,"syntology":null},{"paper":null,"slug":"sample-complexity-of-asynchronous-q-learning","title":"Sample Complexity of Asynchronous Q-Learning: Sharper Analysis and Variance Reduction","date":"2020-06-04","arxiv_id":"2006.03041","n_code_links":0,"syntology":null},{"paper":null,"slug":"mitigating-bias-in-face-recognition-using","title":"Mitigating Bias in Face Recognition Using Skewness-Aware Reinforcement Learning","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-understanding-linear-value","title":"Towards Understanding Cooperative Multi-Agent Q-Learning with Value Factorization","date":"2020-05-31","arxiv_id":"2006.00587","n_code_links":0,"syntology":null},{"paper":null,"slug":"active-measure-reinforcement-learning-for","title":"Active Measure Reinforcement Learning for Observation Cost Minimization","date":"2020-05-26","arxiv_id":"2005.12697","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-charge-rf-energy-harvesting","title":"Learning to Charge RF-Energy Harvesting Devices in WiFi Networks","date":"2020-05-25","arxiv_id":"2005.12022","n_code_links":0,"syntology":null},{"paper":null,"slug":"should-artificial-agents-ask-for-help-in","title":"Should artificial agents ask for help in human-robot collaborative problem-solving?","date":"2020-05-25","arxiv_id":"2006.00882","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-reinforcement-learning-based-decision","title":"A reinforcement learning based decision support system in textile manufacturing process","date":"2020-05-20","arxiv_id":"2005.09867","n_code_links":0,"syntology":null},{"paper":null,"slug":"prototypical-q-networks-for-automatic","title":"Prototypical Q Networks for Automatic Conversational Diagnosis and Few-Shot New Disease Adaption","date":"2020-05-19","arxiv_id":"2005.11153","n_code_links":0,"syntology":null},{"paper":null,"slug":"basal-glucose-control-in-type-1-diabetes","title":"Basal Glucose Control in Type 1 Diabetes using Deep Reinforcement Learning: An In Silico Validation","date":"2020-05-18","arxiv_id":"2005.09059","n_code_links":0,"syntology":null},{"paper":"/paper/local-and-global-explanations-of-agent","slug":"local-and-global-explanations-of-agent","title":"Local and Global Explanations of Agent Behavior: Integrating Strategy Summaries with Saliency Maps","date":"2020-05-18","arxiv_id":"2005.08874","n_code_links":1,"syntology":{"ran":9,"of":15,"n_ran_checked":9,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["HuTobias/HIGHLIGHTS-LRP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-deep-q-learning-genetic-algorithms-based","title":"A Deep Q-learning/genetic Algorithms Based Novel Methodology For Optimizing Covid-19 Pandemic Government Actions","date":"2020-05-15","arxiv_id":"2005.07656","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deep-reinforcement-learning-approach-to-2","title":"A Deep Reinforcement Learning Approach to Efficient Drone Mobility Support","date":"2020-05-11","arxiv_id":"2005.05229","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-fpga-based-on-device-reinforcement","title":"An FPGA-Based On-Device Reinforcement Learning Approach using Online Sequential Learning","date":"2020-05-10","arxiv_id":"2005.04646","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-thermostatically","title":"Reinforcement Learning for Thermostatically Controlled Loads Control using Modelica and Python","date":"2020-05-09","arxiv_id":"2005.04444","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimal-beam-association-for-high-mobility","title":"Optimal Beam Association for High Mobility mmWave Vehicular Networks: Lightweight Parallel Reinforcement Learning Approach","date":"2020-05-02","arxiv_id":"2005.00694","n_code_links":0,"syntology":null},{"paper":null,"slug":"implementing-inductive-bias-for-different","title":"Implementing Inductive bias for different navigation tasks through diverse RNN attrractors","date":"2020-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-efficient-parameter-server","title":"Learning Efficient Parameter Server Synchronization Policies for Distributed SGD","date":"2020-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"whittle-index-based-q-learning-for-restless","title":"Whittle index based Q-learning for restless bandits with average reward","date":"2020-04-29","arxiv_id":"2004.14427","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-dialog-policies-from-weak","title":"Learning Dialog Policies from Weak Demonstrations","date":"2020-04-23","arxiv_id":"2004.11054","n_code_links":0,"syntology":null},{"paper":"/paper/spatial-action-maps-for-mobile-manipulation","slug":"spatial-action-maps-for-mobile-manipulation","title":"Spatial Action Maps for Mobile Manipulation","date":"2020-04-20","arxiv_id":"2004.09141","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":6,"n_instrument":2,"unverified":0,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jimmyyhwu/spatial-action-maps"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-reinforcement-learning-for-adaptive-1","title":"Deep Reinforcement Learning for Adaptive Learning Systems","date":"2020-04-17","arxiv_id":"2004.08410","n_code_links":0,"syntology":null},{"paper":null,"slug":"show-us-the-way-learning-to-manage-dialog","title":"Show Us the Way: Learning to Manage Dialog from Demonstrations","date":"2020-04-17","arxiv_id":"2004.08114","n_code_links":0,"syntology":null},{"paper":null,"slug":"k-spin-hamiltonian-for-quantum-resolvable","title":"K-spin Hamiltonian for quantum-resolvable Markov decision processes","date":"2020-04-13","arxiv_id":"2004.06040","n_code_links":0,"syntology":null},{"paper":null,"slug":"risk-aware-high-level-decisions-for-automated","title":"Risk-Aware High-level Decisions for Automated Driving at Occluded Intersections with Reinforcement Learning","date":"2020-04-09","arxiv_id":"2004.04450","n_code_links":0,"syntology":null},{"paper":"/paper/an-application-of-deep-reinforcement-learning","slug":"an-application-of-deep-reinforcement-learning","title":"An Application of Deep Reinforcement Learning to Algorithmic Trading","date":"2020-04-07","arxiv_id":"2004.06627","n_code_links":1,"syntology":null},{"paper":null,"slug":"uniform-state-abstraction-for-reinforcement","title":"Uniform State Abstraction For Reinforcement Learning","date":"2020-04-06","arxiv_id":"2004.02919","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-learning-of-text-adventure-games","title":"Zero-Shot Learning of Text Adventure Games with Sentence-Level Semantics","date":"2020-04-06","arxiv_id":"2004.02986","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-agent-reinforcement-learning-for-3","title":"Multi-agent Reinforcement Learning for Resource Allocation in IoT networks with Edge Computing","date":"2020-04-05","arxiv_id":"2004.02315","n_code_links":0,"syntology":null},{"paper":null,"slug":"minimizing-age-of-information-for-fog","title":"Minimizing Age-of-Information for Fog Computing-supported Vehicular Networks with Deep Q-learning","date":"2020-04-04","arxiv_id":"2004.04640","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-mixed-integer","title":"Reinforcement Learning for Mixed-Integer Problems Based on MPC","date":"2020-04-03","arxiv_id":"2004.01430","n_code_links":0,"syntology":null},{"paper":null,"slug":"statistically-model-checking-pctl","title":"Statistically Model Checking PCTL Specifications on Markov Decision Processes via Reinforcement Learning","date":"2020-04-01","arxiv_id":"2004.00273","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhanced-rolling-horizon-evolution-algorithm","title":"Enhanced Rolling Horizon Evolution Algorithm with Opponent Model Learning: Results for the Fighting Game AI Competition","date":"2020-03-31","arxiv_id":"2003.13949","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-q-learning","title":"Robust Q-learning","date":"2020-03-27","arxiv_id":"2003.12427","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-recent-advancements-in-model-based-deep-1","title":"Importance of using appropriate baselines for evaluation of data-efficiency in deep reinforcement learning for Atari","date":"2020-03-23","arxiv_id":"2003.10181","n_code_links":0,"syntology":null},{"paper":"/paper/using-deep-reinforcement-learning-methods-for","slug":"using-deep-reinforcement-learning-methods-for","title":"Using Deep Reinforcement Learning Methods for Autonomous Vessels in 2D Environments","date":"2020-03-23","arxiv_id":"2003.10249","n_code_links":1,"syntology":null},{"paper":null,"slug":"distributed-reinforcement-learning-for-1","title":"Distributed Reinforcement Learning for Cooperative Multi-Robot Object Manipulation","date":"2020-03-21","arxiv_id":"2003.09540","n_code_links":0,"syntology":null},{"paper":"/paper/flapai-bird-training-an-agent-to-play-flappy","slug":"flapai-bird-training-an-agent-to-play-flappy","title":"FlapAI Bird: Training an Agent to Play Flappy Bird Using Reinforcement Learning Techniques","date":"2020-03-21","arxiv_id":"2003.09579","n_code_links":2,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-with-weighted-q","title":"Deep Reinforcement Learning with Weighted Q-Learning","date":"2020-03-20","arxiv_id":"2003.09280","n_code_links":0,"syntology":null},{"paper":null,"slug":"interpretable-multi-time-scale-constraints-in","title":"Deep Constrained Q-learning","date":"2020-03-20","arxiv_id":"2003.09398","n_code_links":0,"syntology":null},{"paper":"/paper/robust-deep-reinforcement-learning-against","slug":"robust-deep-reinforcement-learning-against","title":"Robust Deep Reinforcement Learning against Adversarial Perturbations on State Observations","date":"2020-03-19","arxiv_id":"2003.08938","n_code_links":4,"syntology":null},{"paper":"/paper/simultaneous-navigation-and-radio-mapping-for","slug":"simultaneous-navigation-and-radio-mapping-for","title":"Simultaneous Navigation and Radio Mapping for Cellular-Connected UAV with Deep Reinforcement Learning","date":"2020-03-17","arxiv_id":"2003.07574","n_code_links":1,"syntology":null},{"paper":"/paper/discor-corrective-feedback-in-reinforcement","slug":"discor-corrective-feedback-in-reinforcement","title":"DisCor: Corrective Feedback in Reinforcement Learning via Distribution Correction","date":"2020-03-16","arxiv_id":"2003.07305","n_code_links":4,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"active-perception-and-representation-for","title":"Active Perception and Representation for Robotic Manipulation","date":"2020-03-15","arxiv_id":"2003.06734","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-general-framework-for-learning-mean-field","title":"A General Framework for Learning Mean-Field Games","date":"2020-03-13","arxiv_id":"2003.06069","n_code_links":0,"syntology":null},{"paper":null,"slug":"application-of-deep-q-network-in-portfolio","title":"Application of Deep Q-Network in Portfolio Management","date":"2020-03-13","arxiv_id":"2003.06365","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-free-algorithm-and-regret-analysis-for","title":"Provably Efficient Model-Free Algorithm for MDPs with Peak Constraints","date":"2020-03-11","arxiv_id":"2003.05555","n_code_links":0,"syntology":null},{"paper":null,"slug":"behavior-planning-for-connected-autonomous","title":"A Multi-Agent Reinforcement Learning Approach For Safe and Efficient Behavior Planning Of Connected Autonomous Vehicles","date":"2020-03-09","arxiv_id":"2003.04371","n_code_links":0,"syntology":null},{"paper":null,"slug":"software-level-accuracy-using-stochastic","title":"Software-Level Accuracy Using Stochastic Computing With Charge-Trap-Flash Based Weight Matrix","date":"2020-03-09","arxiv_id":"2004.11120","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-experience-replay","title":"Dynamic Experience Replay","date":"2020-03-04","arxiv_id":"2003.02372","n_code_links":0,"syntology":null},{"paper":"/paper/contention-window-optimization-in-ieee","slug":"contention-window-optimization-in-ieee","title":"Contention Window Optimization in IEEE 802.11ax Networks with Deep Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01492","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-supervised-object-level-deep","title":"Relevance-Guided Modeling of Object Dynamics for Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01384","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-in-flipit","title":"Deep Reinforcement Learning for FlipIt Security Game","date":"2020-02-28","arxiv_id":"2002.12909","n_code_links":0,"syntology":null},{"paper":"/paper/conqur-mitigating-delusional-bias-in-deep-q-1","slug":"conqur-mitigating-delusional-bias-in-deep-q-1","title":"ConQUR: Mitigating Delusional Bias in Deep Q-learning","date":"2020-02-27","arxiv_id":"2002.12399","n_code_links":1,"syntology":null},{"paper":"/paper/optimistic-exploration-even-with-a-1","slug":"optimistic-exploration-even-with-a-1","title":"Optimistic Exploration even with a Pessimistic Initialisation","date":"2020-02-26","arxiv_id":"2002.12174","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["oxwhirl/opiq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"g-learner-and-girl-goal-based-wealth","title":"G-Learner and GIRL: Goal Based Wealth Management with Reinforcement Learning","date":"2020-02-25","arxiv_id":"2002.10990","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-double-q-learning-approach-for-navigation","title":"A Double Q-Learning Approach for Navigation of Aerial Vehicles with Connectivity Constraint","date":"2020-02-24","arxiv_id":"2002.10563","n_code_links":0,"syntology":null},{"paper":null,"slug":"millimeter-wave-communications-with-an","title":"Millimeter Wave Communications with an Intelligent Reflector: Performance Optimization and Distributional Reinforcement Learning","date":"2020-02-24","arxiv_id":"2002.10572","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-learning-with-uniformly-bounded-variance","title":"Q-learning with Uniformly Bounded Variance: Large Discounting is Not a Barrier to Fast Learning","date":"2020-02-24","arxiv_id":"2002.10301","n_code_links":0,"syntology":null}],"record_sha256":"5056f704999fcbdfef68fc1d8e9b6f9f968afaf06eb0093631cd3e72927888e8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}