{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/8","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":8,"pages_in_order":18,"rows_per_page":100,"rows":[701,800],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning/papers/7","next":"/method/q-learning/papers/9","papers":[{"paper":null,"slug":"partial-counterfactual-identification-for","title":"Partial Counterfactual Identification for Infinite Horizon Partially Observable Markov Decision Process","date":"2022-08-31","arxiv_id":"2209.00137","n_code_links":0,"syntology":null},{"paper":null,"slug":"prospect-theory-inspired-automated-p2p-energy","title":"Prospect Theory-inspired Automated P2P Energy Trading with Q-learning-based Dynamic Pricing","date":"2022-08-26","arxiv_id":"2208.12777","n_code_links":0,"syntology":null},{"paper":null,"slug":"recurrent-neural-network-based-anti-jamming","title":"Recurrent Neural Network-based Anti-jamming Framework for Defense Against Multiple Jamming Policies","date":"2022-08-19","arxiv_id":"2208.09518","n_code_links":0,"syntology":null},{"paper":"/paper/diffusion-policies-as-an-expressive-policy","slug":"diffusion-policies-as-an-expressive-policy","title":"Diffusion Policies as an Expressive Policy Class for Offline Reinforcement Learning","date":"2022-08-12","arxiv_id":"2208.06193","n_code_links":3,"syntology":{"ran":11,"of":18,"n_ran_checked":11,"n_instrument":0,"unverified":7,"pointer_only":10,"phrase":"11 ran (of which 5 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["zhendong-wang/diffusion-policies-for-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/disentangled-modeling-of-domain-and-relevance","slug":"disentangled-modeling-of-domain-and-relevance","title":"Disentangled Modeling of Domain and Relevance for Adaptable Dense Retrieval","date":"2022-08-11","arxiv_id":"2208.05753","n_code_links":1,"syntology":null},{"paper":null,"slug":"prediction-based-hybrid-slicing-framework-for","title":"Prediction-based Hybrid Slicing Framework for Service Level Agreement Guarantee in Mobility Scenarios: A Deep Learning Approach","date":"2022-08-06","arxiv_id":"2208.03460","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-joint-v2i-network","title":"Reinforcement Learning for Joint V2I Network Selection and Autonomous Driving Policies","date":"2022-08-03","arxiv_id":"2208.02249","n_code_links":0,"syntology":null},{"paper":null,"slug":"digital-twin-assisted-efficient-reinforcement","title":"Digital Twin-Assisted Efficient Reinforcement Learning for Edge Task Scheduling","date":"2022-08-02","arxiv_id":"2208.01781","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-maintenance-planning-framework-using-online","title":"A Maintenance Planning Framework using Online and Offline Deep Reinforcement Learning","date":"2022-08-01","arxiv_id":"2208.00808","n_code_links":0,"syntology":null},{"paper":"/paper/off-policy-correction-for-actor-critic","slug":"off-policy-correction-for-actor-critic","title":"Mitigating Off-Policy Bias in Actor-Critic Methods with One-Step Q-learning: A Novel Correction Approach","date":"2022-08-01","arxiv_id":"2208.00755","n_code_links":1,"syntology":null},{"paper":null,"slug":"playing-a-2d-game-indefinitely-using-neat-and","title":"Playing a 2D Game Indefinitely using NEAT and Reinforcement Learning","date":"2022-07-28","arxiv_id":"2207.14140","n_code_links":0,"syntology":null},{"paper":null,"slug":"structural-similarity-for-improved-transfer","title":"Structural Similarity for Improved Transfer in Reinforcement Learning","date":"2022-07-27","arxiv_id":"2207.13813","n_code_links":0,"syntology":null},{"paper":null,"slug":"finite-time-analysis-of-asynchronous-q-1","title":"Finite-Time Analysis of Asynchronous Q-learning under Diminishing Step-Size from Control-Theoretic View","date":"2022-07-25","arxiv_id":"2207.12217","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-source-aoi-constrained-resource","title":"Multi-Source AoI-Constrained Resource Minimization under HARQ: Heterogeneous Sampling Processes","date":"2022-07-19","arxiv_id":"2207.08996","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-decentralizing-federated-reinforcement","title":"On Decentralizing Federated Reinforcement Learning in Multi-Robot Scenarios","date":"2022-07-19","arxiv_id":"2207.09372","n_code_links":0,"syntology":null},{"paper":"/paper/a-deep-reinforcement-learning-approach-for-13","slug":"a-deep-reinforcement-learning-approach-for-13","title":"A Deep Reinforcement Learning Approach for Finding Non-Exploitable Strategies in Two-Player Atari Games","date":"2022-07-18","arxiv_id":"2207.08894","n_code_links":2,"syntology":{"ran":5,"of":10,"n_ran_checked":5,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["quantumiracle/mars","quantumiracle/nash-dqn"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"boolean-decision-rules-for-reinforcement","title":"Boolean Decision Rules for Reinforcement Learning Policy Summarisation","date":"2022-07-18","arxiv_id":"2207.08651","n_code_links":0,"syntology":null},{"paper":null,"slug":"ddpg-learning-for-aerial-ris-assisted-mu-miso","title":"DDPG Learning for Aerial RIS-Assisted MU-MISO Communications","date":"2022-07-13","arxiv_id":"2207.06064","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-objective-optimization-of-notifications","title":"Multi-objective Optimization of Notifications Using Offline Reinforcement Learning","date":"2022-07-07","arxiv_id":"2207.03029","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-learning-in-continuous-time","title":"q-Learning in Continuous Time","date":"2022-07-02","arxiv_id":"2207.00713","n_code_links":0,"syntology":null},{"paper":null,"slug":"interactive-learning-from-natural-language","title":"Interactive Learning from Natural Language and Demonstrations using Signal Temporal Logic","date":"2022-07-01","arxiv_id":"2207.00627","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-with-swin","slug":"deep-reinforcement-learning-with-swin","title":"Deep Reinforcement Learning with Swin Transformers","date":"2022-06-30","arxiv_id":"2206.15269","n_code_links":1,"syntology":null},{"paper":"/paper/on-the-learning-and-learnablity-of","slug":"on-the-learning-and-learnablity-of","title":"On the Learning and Learnability of Quasimetrics","date":"2022-06-30","arxiv_id":"2206.15478","n_code_links":2,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ssnl/poisson_quasimetric_embedding"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"recursive-reinforcement-learning","title":"Recursive Reinforcement Learning","date":"2022-06-23","arxiv_id":"2206.11430","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-under-partial","title":"Reinforcement Learning under Partial Observability Guided by Learned Environment Models","date":"2022-06-23","arxiv_id":"2206.11708","n_code_links":0,"syntology":null},{"paper":"/paper/multi-agent-car-parking-using-reinforcement","slug":"multi-agent-car-parking-using-reinforcement","title":"Multi-Agent Car Parking using Reinforcement Learning","date":"2022-06-22","arxiv_id":"2206.13338","n_code_links":1,"syntology":null},{"paper":null,"slug":"federated-reinforcement-learning-linear","title":"Federated Stochastic Approximation under Markov Noise and Heterogeneity: Applications in Reinforcement Learning","date":"2022-06-21","arxiv_id":"2206.10185","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-integration-of-machine-learning-into","title":"The Integration of Machine Learning into Automated Test Generation: A Systematic Mapping Study","date":"2022-06-21","arxiv_id":"2206.10210","n_code_links":0,"syntology":null},{"paper":"/paper/dna-proximal-policy-optimization-with-a-dual","slug":"dna-proximal-policy-optimization-with-a-dual","title":"DNA: Proximal Policy Optimization with a Dual Network Architecture","date":"2022-06-20","arxiv_id":"2206.10027","n_code_links":1,"syntology":null},{"paper":"/paper/maser-multi-agent-reinforcement-learning-with","slug":"maser-multi-agent-reinforcement-learning-with","title":"MASER: Multi-Agent Reinforcement Learning with Subgoals Generated from Experience Replay Buffer","date":"2022-06-20","arxiv_id":"2206.10607","n_code_links":1,"syntology":null},{"paper":"/paper/sampling-efficient-deep-reinforcement","slug":"sampling-efficient-deep-reinforcement","title":"Sampling Efficient Deep Reinforcement Learning through Preference-Guided Stochastic Exploration","date":"2022-06-20","arxiv_id":"2206.09627","n_code_links":1,"syntology":null},{"paper":null,"slug":"visual-radial-basis-q-network","title":"Visual Radial Basis Q-Network","date":"2022-06-14","arxiv_id":"2206.06712","n_code_links":0,"syntology":null},{"paper":null,"slug":"rl-ea-a-reinforcement-learning-based","title":"RL-GA: A Reinforcement Learning-Based Genetic Algorithm for Electromagnetic Detection Satellite Scheduling Problem","date":"2022-06-12","arxiv_id":"2206.05694","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-optimization-method-assisted-ensemble-deep","title":"An Optimization Method-Assisted Ensemble Deep Reinforcement Learning Algorithm to Solve Unit Commitment Problems","date":"2022-06-09","arxiv_id":"2206.04249","n_code_links":0,"syntology":null},{"paper":"/paper/mildly-conservative-q-learning-for-offline","slug":"mildly-conservative-q-learning-for-offline","title":"Mildly Conservative Q-Learning for Offline Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04745","n_code_links":3,"syntology":{"ran":13,"of":16,"n_ran_checked":13,"n_instrument":0,"unverified":3,"pointer_only":11,"phrase":"13 ran (of which 6 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["dmksjfl/mcq"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"paper":null,"slug":"concentration-bounds-for-ssp-q-learning-for","title":"Concentration bounds for SSP Q-learning for average cost MDPs","date":"2022-06-07","arxiv_id":"2206.03328","n_code_links":0,"syntology":null},{"paper":"/paper/deeptpi-test-point-insertion-with-deep","slug":"deeptpi-test-point-insertion-with-deep","title":"DeepTPI: Test Point Insertion with Deep Reinforcement Learning","date":"2022-06-07","arxiv_id":"2206.06975","n_code_links":1,"syntology":null},{"paper":null,"slug":"balancing-profit-risk-and-sustainability-for","title":"Balancing Profit, Risk, and Sustainability for Portfolio Management","date":"2022-06-06","arxiv_id":"2207.02134","n_code_links":0,"syntology":null},{"paper":null,"slug":"goal-space-planning-with-subgoal-models","title":"Goal-Space Planning with Subgoal Models","date":"2022-06-06","arxiv_id":"2206.02902","n_code_links":0,"syntology":null},{"paper":"/paper/offline-rl-for-natural-language-generation","slug":"offline-rl-for-natural-language-generation","title":"Offline RL for Natural Language Generation with Implicit Language Q Learning","date":"2022-06-05","arxiv_id":"2206.11871","n_code_links":2,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"the-phenomenon-of-policy-churn","title":"The Phenomenon of Policy Churn","date":"2022-06-01","arxiv_id":"2206.00730","n_code_links":0,"syntology":null},{"paper":null,"slug":"designing-rewards-for-fast-learning","title":"Designing Rewards for Fast Learning","date":"2022-05-30","arxiv_id":"2205.15400","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-reward-poisoning-attacks-on-online","slug":"efficient-reward-poisoning-attacks-on-online","title":"Efficient Reward Poisoning Attacks on Online Deep Reinforcement Learning","date":"2022-05-30","arxiv_id":"2205.14842","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yinglunxu/reward_poisoning_attack_drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-reinforcement-learning-for-distributed-1","title":"Deep Reinforcement Learning for Distributed and Uncoordinated Cognitive Radios Resource Allocation","date":"2022-05-27","arxiv_id":"2205.13944","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-bidding-and-playing-strategies-in","title":"Improving Bidding and Playing Strategies in the Trick-Taking game Wizard using Deep Q-Networks","date":"2022-05-27","arxiv_id":"2205.13834","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-experimental-comparison-between-temporal","title":"An Experimental Comparison Between Temporal Difference and Residual Gradient with Neural Network Approximation","date":"2022-05-25","arxiv_id":"2205.12770","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-for-multi-class","slug":"deep-reinforcement-learning-for-multi-class","title":"Deep Reinforcement Learning for Multi-class Imbalanced Training","date":"2022-05-24","arxiv_id":"2205.12070","n_code_links":1,"syntology":null},{"paper":null,"slug":"multiple-domain-cyberspace-attack-and-defense","title":"Multiple Domain Cyberspace Attack and Defense Game Based on Reward Randomization Reinforcement Learning","date":"2022-05-23","arxiv_id":"2205.10990","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-returns-using-the-hurst-exponent","title":"Optimizing Returns Using the Hurst Exponent and Q Learning on Momentum and Mean Reversion Strategies","date":"2022-05-23","arxiv_id":"2205.11122","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforced-pedestrian-attribute-recognition","title":"Reinforced Pedestrian Attribute Recognition with Group Optimization Reward","date":"2022-05-21","arxiv_id":"2205.14042","n_code_links":0,"syntology":null},{"paper":null,"slug":"long-run-incremental-cost-lric-distribution","title":"Long Run Incremental Cost (LRIC) Distribution Network Pricing in UK, advising China's Distribution Network","date":"2022-05-20","arxiv_id":"2205.09946","n_code_links":0,"syntology":null},{"paper":null,"slug":"parallel-bandit-architecture-based-on-laser","title":"Parallel bandit architecture based on laser chaos for reinforcement learning","date":"2022-05-19","arxiv_id":"2205.09543","n_code_links":0,"syntology":null},{"paper":null,"slug":"enforcing-kl-regularization-in-general","title":"Enforcing KL Regularization in General Tsallis Entropy Reinforcement Learning via Advantage Learning","date":"2022-05-16","arxiv_id":"2205.07885","n_code_links":0,"syntology":null},{"paper":null,"slug":"qhd-a-brain-inspired-hyperdimensional","title":"Efficient Off-Policy Reinforcement Learning via Brain-Inspired Computing","date":"2022-05-14","arxiv_id":"2205.06978","n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-learning-for-context-dependent","title":"Representation Learning for Context-Dependent Decision-Making","date":"2022-05-12","arxiv_id":"2205.05820","n_code_links":0,"syntology":null},{"paper":null,"slug":"characterizing-the-action-generalization-gap","title":"Characterizing the Action-Generalization Gap in Deep Q-Learning","date":"2022-05-11","arxiv_id":"2205.05588","n_code_links":0,"syntology":null},{"paper":null,"slug":"finite-time-analysis-of-constant-step-size-q","title":"Final Iteration Convergence Bound of Q-Learning: Switching System Approach","date":"2022-05-11","arxiv_id":"2205.05455","n_code_links":0,"syntology":null},{"paper":null,"slug":"neuromimetic-linear-systems-resilience-and","title":"Neuromimetic Linear Systems -- Resilience and Learning","date":"2022-05-10","arxiv_id":"2205.05013","n_code_links":0,"syntology":null},{"paper":"/paper/simultaneous-double-q-learning-with","slug":"simultaneous-double-q-learning-with","title":"Simultaneous Double Q-learning with Conservative Advantage Learning for Actor-Critic Methods","date":"2022-05-08","arxiv_id":"2205.03819","n_code_links":1,"syntology":null},{"paper":null,"slug":"chemoreception-and-chemotaxis-of-a-three","title":"Chemoreception and chemotaxis of a three-sphere swimmer","date":"2022-05-05","arxiv_id":"2205.02678","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-value-functions-from-undirected-1","title":"Learning Value Functions from Undirected State-only Experience","date":"2022-04-26","arxiv_id":"2204.12458","n_code_links":0,"syntology":null},{"paper":"/paper/epida-an-easy-plug-in-data-augmentation","slug":"epida-an-easy-plug-in-data-augmentation","title":"EPiDA: An Easy Plug-in Data Augmentation Framework for High Performance Text Classification","date":"2022-04-24","arxiv_id":"2204.11205","n_code_links":1,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["zhaominyiz/epida"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"graph-neural-network-based-agent-in-google","title":"Graph Neural Network based Agent in Google Research Football","date":"2022-04-23","arxiv_id":"2204.11142","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-how-to-interact-with-a-complex","title":"Learning how to Interact with a Complex Interface using Hierarchical Reinforcement Learning","date":"2022-04-21","arxiv_id":"2204.10374","n_code_links":0,"syntology":null},{"paper":null,"slug":"provably-efficient-kernelized-q-learning","title":"Provably Efficient Kernelized Q-Learning","date":"2022-04-21","arxiv_id":"2204.10349","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-learning-of-reward-machines-and","title":"Joint Learning of Reward Machines and Policies in Environments with Partially Known Semantics","date":"2022-04-20","arxiv_id":"2204.11833","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-re-ranking-with-2d-grid-based","title":"Reinforcement Re-ranking with 2D Grid-based Recommendation Panels","date":"2022-04-11","arxiv_id":"2204.04954","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-the-long-term-behaviour-of-deep","title":"Optimizing the Long-Term Behaviour of Deep Reinforcement Learning for Pushing and Grasping","date":"2022-04-07","arxiv_id":"2204.03487","n_code_links":0,"syntology":null},{"paper":"/paper/douzero-improving-doudizhu-ai-by-opponent","slug":"douzero-improving-doudizhu-ai-by-opponent","title":"DouZero+: Improving DouDizhu AI by Opponent Modeling and Coach-guided Learning","date":"2022-04-06","arxiv_id":"2204.02558","n_code_links":1,"syntology":null},{"paper":"/paper/gail-pt-a-generic-intelligent-penetration","slug":"gail-pt-a-generic-intelligent-penetration","title":"GAIL-PT: A Generic Intelligent Penetration Testing Framework with Generative Adversarial Imitation Learning","date":"2022-04-05","arxiv_id":"2204.01975","n_code_links":1,"syntology":null},{"paper":null,"slug":"rem-routing-entropy-minimization-for-capsule","title":"REM: Routing Entropy Minimization for Capsule Networks","date":"2022-04-04","arxiv_id":"2204.01298","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-q-learning-of-global-optimizer-of","title":"Deep Q-learning of global optimizer of multiply model parameters for viscoelastic imaging","date":"2022-04-01","arxiv_id":"2204.01844","n_code_links":0,"syntology":null},{"paper":"/paper/hysteresis-based-rl-robustifying","slug":"hysteresis-based-rl-robustifying","title":"Hysteresis-Based RL: Robustifying Reinforcement Learning-based Control Policies via Hybrid Control","date":"2022-04-01","arxiv_id":"2204.00654","n_code_links":2,"syntology":null},{"paper":null,"slug":"functional-stability-of-discounted-markov","title":"Functional Stability of Discounted Markov Decision Processes Using Economic MPC Dissipativity Theory","date":"2022-03-31","arxiv_id":"2203.16989","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-q-learning-for-solving-elliptic-pdes","title":"Neural Q-learning for solving PDEs","date":"2022-03-31","arxiv_id":"2203.17128","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-the-properties-of-neural","title":"Investigating the Properties of Neural Network Representations in Reinforcement Learning","date":"2022-03-30","arxiv_id":"2203.15955","n_code_links":0,"syntology":null},{"paper":"/paper/topological-experience-replay-1","slug":"topological-experience-replay-1","title":"Topological Experience Replay","date":"2022-03-29","arxiv_id":"2203.15845","n_code_links":1,"syntology":null},{"paper":null,"slug":"merlin-malware-evasion-with-reinforcement","title":"MERLIN -- Malware Evasion with Reinforcement LearnINg","date":"2022-03-24","arxiv_id":"2203.12980","n_code_links":0,"syntology":null},{"paper":null,"slug":"resource-allocation-optimization-using","title":"The state-of-the-art review on resource allocation problem using artificial intelligence methods on various computing paradigms","date":"2022-03-23","arxiv_id":"2203.12315","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-note-on-target-q-learning-for-solving","title":"A Note on Target Q-learning For Solving Finite MDPs with A Generative Oracle","date":"2022-03-22","arxiv_id":"2203.11489","n_code_links":0,"syntology":null},{"paper":"/paper/action-candidate-driven-clipped-double-q","slug":"action-candidate-driven-clipped-double-q","title":"Action Candidate Driven Clipped Double Q-learning for Discrete and Continuous Action Tasks","date":"2022-03-22","arxiv_id":"2203.11526","n_code_links":1,"syntology":null},{"paper":null,"slug":"distributed-learning-for-vehicular-dynamic","title":"Distributed Learning for Vehicular Dynamic Spectrum Access in Autonomous Driving","date":"2022-03-22","arxiv_id":"2204.10179","n_code_links":0,"syntology":null},{"paper":null,"slug":"does-dqn-really-learn-exploring-adversarial","title":"Does DQN really learn? Exploring adversarial training schemes in Pong","date":"2022-03-20","arxiv_id":"2203.10614","n_code_links":0,"syntology":null},{"paper":null,"slug":"infinite-horizon-reach-avoid-zero-sum-games","title":"Infinite-Horizon Reach-Avoid Zero-Sum Games via Deep Reinforcement Learning","date":"2022-03-18","arxiv_id":"2203.10142","n_code_links":0,"syntology":null},{"paper":"/paper/orchestrated-value-mapping-for-reinforcement-1","slug":"orchestrated-value-mapping-for-reinforcement-1","title":"Orchestrated Value Mapping for Reinforcement Learning","date":"2022-03-14","arxiv_id":"2203.07171","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-optimal-control-of","title":"Reinforcement Learning for Optimal Control of a District Cooling Energy Plant","date":"2022-03-14","arxiv_id":"2203.07500","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-efficacy-of-pessimism-in-asynchronous-q","title":"The Efficacy of Pessimism in Asynchronous Q-Learning","date":"2022-03-14","arxiv_id":"2203.07368","n_code_links":0,"syntology":null},{"paper":null,"slug":"random-ensemble-reinforcement-learning-for","title":"Random Ensemble Reinforcement Learning for Traffic Signal Control","date":"2022-03-10","arxiv_id":"2203.05961","n_code_links":0,"syntology":null},{"paper":null,"slug":"graph-based-reinforcement-learning-meets","title":"Graph-based Reinforcement Learning meets Mixed Integer Programs: An application to 3D robot assembly discovery","date":"2022-03-08","arxiv_id":"2203.04120","n_code_links":0,"syntology":null},{"paper":null,"slug":"scalable-multi-agent-reinforcement-learning-1","title":"Scalable multi-agent reinforcement learning for distributed control of residential energy flexibility","date":"2022-03-07","arxiv_id":"2203.03417","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-model-free","title":"Deep Reinforcement Learning based Model-free On-line Dynamic Multi-Microgrid Formation to Enhance Resilience","date":"2022-03-06","arxiv_id":"2203.03030","n_code_links":0,"syntology":null},{"paper":null,"slug":"offline-deep-reinforcement-learning-for","title":"Offline Deep Reinforcement Learning for Dynamic Pricing of Consumer Credit","date":"2022-03-06","arxiv_id":"2203.03003","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-learning-based-framework-for-handling","title":"A Learning Based Framework for Handling Uncertain Lead Times in Multi-Product Inventory Management","date":"2022-03-02","arxiv_id":"2203.00885","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-the-diversity-of-bootstrapped-dqn","title":"Improving the Diversity of Bootstrapped DQN by Replacing Priors With Noise","date":"2022-03-02","arxiv_id":"2203.01004","n_code_links":0,"syntology":null},{"paper":null,"slug":"pessimistic-q-learning-for-offline","title":"Pessimistic Q-Learning for Offline Reinforcement Learning: Towards Optimal Sample Complexity","date":"2022-02-28","arxiv_id":"2202.13890","n_code_links":0,"syntology":null},{"paper":null,"slug":"autonomous-warehouse-robot-using-deep-q","title":"Autonomous Warehouse Robot using Deep Q-Learning","date":"2022-02-21","arxiv_id":"2202.10019","n_code_links":0,"syntology":null},{"paper":null,"slug":"pool-pheromone-inspired-communication","title":"PooL: Pheromone-inspired Communication Framework forLarge Scale Multi-Agent Reinforcement Learning","date":"2022-02-20","arxiv_id":"2202.09722","n_code_links":0,"syntology":null},{"paper":null,"slug":"retrieval-augmented-reinforcement-learning-1","title":"Retrieval-Augmented Reinforcement Learning","date":"2022-02-17","arxiv_id":"2202.08417","n_code_links":0,"syntology":null},{"paper":"/paper/goal-recognition-as-reinforcement-learning","slug":"goal-recognition-as-reinforcement-learning","title":"Goal Recognition as Reinforcement Learning","date":"2022-02-13","arxiv_id":"2202.06356","n_code_links":1,"syntology":null},{"paper":null,"slug":"regularized-q-learning","title":"Regularized Q-learning","date":"2022-02-11","arxiv_id":"2202.05404","n_code_links":0,"syntology":null}],"record_sha256":"4f0fb0bbc56c7a31dc16ef4e35176bf131d13ed2bc725ea0a97501bb9b0b287d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}