{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/7","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":7,"pages_in_order":18,"rows_per_page":100,"rows":[601,700],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning/papers/6","next":"/method/q-learning/papers/8","papers":[{"paper":null,"slug":"macoptions-multi-agent-learning-with","title":"MACOptions: Multi-Agent Learning with Centralized Controller and Options Framework","date":"2023-02-07","arxiv_id":"2302.03800","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-traffic-light-1","title":"Deep Reinforcement Learning for Traffic Light Control in Intelligent Transportation Systems","date":"2023-02-04","arxiv_id":"2302.03669","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-policy-gradient-by-estimating","title":"Accelerating Policy Gradient by Estimating Value Function from Prior Computation in Deep Reinforcement Learning","date":"2023-02-02","arxiv_id":"2302.01399","n_code_links":0,"syntology":null},{"paper":null,"slug":"best-possible-q-learning","title":"Best Possible Q-Learning","date":"2023-02-02","arxiv_id":"2302.01188","n_code_links":0,"syntology":null},{"paper":null,"slug":"diversity-through-exclusion-dte-niche","title":"Diversity Through Exclusion (DTE): Niche Identification for Reinforcement Learning through Value-Decomposition","date":"2023-02-02","arxiv_id":"2302.01180","n_code_links":0,"syntology":null},{"paper":null,"slug":"sample-complexity-of-kernel-based-q-learning","title":"Sample Complexity of Kernel-Based Q-Learning","date":"2023-02-01","arxiv_id":"2302.00727","n_code_links":0,"syntology":null},{"paper":null,"slug":"sample-efficient-deep-reinforcement-learning-3","title":"Sample Efficient Deep Reinforcement Learning via Local Planning","date":"2023-01-29","arxiv_id":"2301.12579","n_code_links":0,"syntology":null},{"paper":null,"slug":"analyzing-robustness-of-the-deep","title":"Analyzing Robustness of the Deep Reinforcement Learning Algorithm in Ramp Metering Applications Considering False Data Injection Attack and Defense","date":"2023-01-28","arxiv_id":"2301.12036","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-impact-of-surplus-sharing-on-the-outcomes","title":"The impact of surplus sharing on the outcomes of specific investments under negotiated transfer pricing: An agent-based simulation with fuzzy Q-learning agents","date":"2023-01-28","arxiv_id":"2301.12255","n_code_links":0,"syntology":null},{"paper":null,"slug":"single-trajectory-distributionally-robust","title":"Single-Trajectory Distributionally Robust Reinforcement Learning","date":"2023-01-27","arxiv_id":"2301.11721","n_code_links":0,"syntology":null},{"paper":null,"slug":"fedhql-federated-heterogeneous-q-learning","title":"FedHQL: Federated Heterogeneous Q-Learning","date":"2023-01-26","arxiv_id":"2301.11135","n_code_links":0,"syntology":null},{"paper":"/paper/learning-from-multiple-independent-advisors","slug":"learning-from-multiple-independent-advisors","title":"Learning from Multiple Independent Advisors in Multi-agent Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.11153","n_code_links":1,"syntology":null},{"paper":null,"slug":"asymptotic-convergence-and-performance-of","title":"Asymptotic Convergence and Performance of Multi-Agent Q-Learning Dynamics","date":"2023-01-23","arxiv_id":"2301.09619","n_code_links":0,"syntology":null},{"paper":null,"slug":"asynchronous-deep-double-duelling-q-learning","title":"Asynchronous Deep Double Duelling Q-Learning for Trading-Signal Execution in Limit Order Book Markets","date":"2023-01-20","arxiv_id":"2301.08688","n_code_links":0,"syntology":null},{"paper":"/paper/modeling-moral-choices-in-social-dilemmas","slug":"modeling-moral-choices-in-social-dilemmas","title":"Modeling Moral Choices in Social Dilemmas with Multi-Agent Reinforcement Learning","date":"2023-01-20","arxiv_id":"2301.08491","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["liza-tennant/moral_choice_dyadic","Liza-Tennant/modeling_moral_choice_dyadic"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"risk-averse-reinforcement-learning-via","title":"Risk-Averse Reinforcement Learning via Dynamic Time-Consistent Risk Measures","date":"2023-01-14","arxiv_id":"2301.05981","n_code_links":0,"syntology":null},{"paper":null,"slug":"decentralized-model-free-reinforcement","title":"Decentralized model-free reinforcement learning in stochastic games with average-reward objective","date":"2023-01-13","arxiv_id":"2301.05630","n_code_links":0,"syntology":null},{"paper":null,"slug":"hierarchical-deep-q-learning-based-handover","title":"Hierarchical Deep Q-Learning Based Handover in Wireless Networks with Dual Connectivity","date":"2023-01-13","arxiv_id":"2301.05391","n_code_links":0,"syntology":null},{"paper":"/paper/transfqmix-transformers-for-leveraging-the","slug":"transfqmix-transformers-for-leveraging-the","title":"TransfQMix: Transformers for Leveraging the Graph Structure of Multi-Agent Reinforcement Learning Problems","date":"2023-01-13","arxiv_id":"2301.05334","n_code_links":1,"syntology":null},{"paper":"/paper/schlably-a-python-framework-for-deep","slug":"schlably-a-python-framework-for-deep","title":"schlably: A Python Framework for Deep Reinforcement Learning Based Scheduling Experiments","date":"2023-01-10","arxiv_id":"2301.04182","n_code_links":1,"syntology":null},{"paper":null,"slug":"tuning-path-tracking-controllers-for","title":"Tuning Path Tracking Controllers for Autonomous Cars Using Reinforcement Learning","date":"2023-01-09","arxiv_id":"2301.03363","n_code_links":0,"syntology":null},{"paper":null,"slug":"xdqn-inherently-interpretable-dqn-through","title":"XDQN: Inherently Interpretable DQN through Mimicking","date":"2023-01-08","arxiv_id":"2301.03043","n_code_links":0,"syntology":null},{"paper":"/paper/extreme-q-learning-maxent-rl-without-entropy","slug":"extreme-q-learning-maxent-rl-without-entropy","title":"Extreme Q-Learning: MaxEnt RL without Entropy","date":"2023-01-05","arxiv_id":"2301.02328","n_code_links":4,"syntology":{"ran":8,"of":13,"n_ran_checked":5,"n_instrument":3,"unverified":5,"pointer_only":7,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":"/paper/training-a-deep-q-learning-agent-inside-a","slug":"training-a-deep-q-learning-agent-inside-a","title":"Learning a Generic Value-Selection Heuristic Inside a Constraint Programming Solver","date":"2023-01-05","arxiv_id":"2301.01913","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-spectral-q-learning-with-application-to","title":"Deep Spectral Q-learning with Application to Mobile Health","date":"2023-01-03","arxiv_id":"2301.00927","n_code_links":0,"syntology":null},{"paper":null,"slug":"hierarchical-deep-reinforcement-learning-for","title":"Hierarchical Deep Reinforcement Learning for Age-of-Information Minimization in IRS-aided and Wireless-powered Wireless Networks","date":"2022-12-27","arxiv_id":"2212.13390","n_code_links":0,"syntology":null},{"paper":null,"slug":"decoding-surface-codes-with-deep","title":"Decoding surface codes with deep reinforcement learning and probabilistic policy reuse","date":"2022-12-22","arxiv_id":"2212.11890","n_code_links":0,"syntology":null},{"paper":"/paper/control-of-continuous-quantum-systems-with","slug":"control-of-continuous-quantum-systems-with","title":"Control of Continuous Quantum Systems with Many Degrees of Freedom based on Convergent Reinforcement Learning","date":"2022-12-21","arxiv_id":"2212.10705","n_code_links":1,"syntology":null},{"paper":null,"slug":"neighboring-state-based-rl-exploration","title":"Neighboring state-based RL Exploration","date":"2022-12-21","arxiv_id":"2212.10712","n_code_links":0,"syntology":null},{"paper":null,"slug":"taming-lagrangian-chaos-with-multi-objective","title":"Taming Lagrangian Chaos with Multi-Objective Reinforcement Learning","date":"2022-12-19","arxiv_id":"2212.09612","n_code_links":0,"syntology":null},{"paper":null,"slug":"offline-robot-reinforcement-learning-with","title":"Offline Robot Reinforcement Learning with Uncertainty-Guided Human Expert Sampling","date":"2022-12-16","arxiv_id":"2212.08232","n_code_links":0,"syntology":null},{"paper":null,"slug":"frugal-reinforcement-based-active-learning","title":"Frugal Reinforcement-based Active Learning","date":"2022-12-09","arxiv_id":"2212.04868","n_code_links":0,"syntology":null},{"paper":null,"slug":"palmer-perception-action-loop-with-memory-for","title":"PALMER: Perception-Action Loop with Memory for Long-Horizon Planning","date":"2022-12-08","arxiv_id":"2212.04581","n_code_links":0,"syntology":null},{"paper":"/paper/policy-transfer-via-enhanced-action-space","slug":"policy-transfer-via-enhanced-action-space","title":"EASpace: Enhanced Action Space for Policy Transfer","date":"2022-12-07","arxiv_id":"2212.03540","n_code_links":1,"syntology":null},{"paper":"/paper/a-machine-with-short-term-episodic-and","slug":"a-machine-with-short-term-episodic-and","title":"A Machine with Short-Term, Episodic, and Semantic Memory Systems","date":"2022-12-05","arxiv_id":"2212.02098","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["humemai/agent-room-env-v1"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/automata-learning-meets-shielding","slug":"automata-learning-meets-shielding","title":"Automata Learning meets Shielding","date":"2022-12-04","arxiv_id":"2212.01838","n_code_links":1,"syntology":null},{"paper":null,"slug":"automatic-discovery-of-multi-perspective","title":"Automatic Discovery of Multi-perspective Process Model using Reinforcement Learning","date":"2022-11-30","arxiv_id":"2211.16687","n_code_links":0,"syntology":null},{"paper":"/paper/welfare-and-fairness-in-multi-objective","slug":"welfare-and-fairness-in-multi-objective","title":"Welfare and Fairness in Multi-objective Reinforcement Learning","date":"2022-11-30","arxiv_id":"2212.01382","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["MuhangTian/Fair-MORL-AAMAS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/ace-cooperative-multi-agent-q-learning-with","slug":"ace-cooperative-multi-agent-q-learning-with","title":"ACE: Cooperative Multi-agent Q-learning with Bidirectional Action-Dependency","date":"2022-11-29","arxiv_id":"2211.16068","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["opendilab/ace"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"airfoil-shape-optimization-using-deep-q","title":"Airfoil Shape Optimization using Deep Q-Network","date":"2022-11-29","arxiv_id":"2211.17189","n_code_links":0,"syntology":null},{"paper":null,"slug":"offline-q-learning-on-diverse-multi-task-data","title":"Offline Q-Learning on Diverse Multi-Task Data Both Scales And Generalizes","date":"2022-11-28","arxiv_id":"2211.15144","n_code_links":0,"syntology":null},{"paper":null,"slug":"qlammp-a-q-learning-agent-for-optimizing-fees","title":"QLAMMP: A Q-Learning Agent for Optimizing Fees on Automated Market Making Protocols","date":"2022-11-28","arxiv_id":"2211.14977","n_code_links":0,"syntology":null},{"paper":"/paper/applying-deep-reinforcement-learning-to-the","slug":"applying-deep-reinforcement-learning-to-the","title":"Applying Deep Reinforcement Learning to the HP Model for Protein Structure Prediction","date":"2022-11-27","arxiv_id":"2211.14939","n_code_links":1,"syntology":{"ran":11,"of":13,"n_ran_checked":11,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["compsoftmatterbiophysics-cityu-hk/applying-drl-to-hp-model-for-protein-structure-prediction"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"reinforcement-causal-structure-learning-on","title":"Reinforcement Causal Structure Learning on Order Graph","date":"2022-11-22","arxiv_id":"2211.12151","n_code_links":0,"syntology":null},{"paper":"/paper/examining-policy-entropy-of-reinforcement","slug":"examining-policy-entropy-of-reinforcement","title":"Examining Policy Entropy of Reinforcement Learning Agents for Personalization Tasks","date":"2022-11-21","arxiv_id":"2211.11869","n_code_links":1,"syntology":null},{"paper":null,"slug":"simultaneously-updating-all-persistence","title":"Simultaneously Updating All Persistence Values in Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11620","n_code_links":0,"syntology":null},{"paper":null,"slug":"analysis-of-reinforcement-learning-schemes","title":"Analysis of Reinforcement Learning Schemes for Trajectory Optimization of an Aerial Radio Unit","date":"2022-11-18","arxiv_id":"2211.10524","n_code_links":0,"syntology":null},{"paper":null,"slug":"credit-cognisant-reinforcement-learning-for","title":"Credit-cognisant reinforcement learning for multi-agent cooperation","date":"2022-11-18","arxiv_id":"2211.10100","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-reinforcement-learning-approach-for-process","title":"A Reinforcement Learning Approach for Process Parameter Optimization in Additive Manufacturing","date":"2022-11-17","arxiv_id":"2211.09545","n_code_links":0,"syntology":null},{"paper":null,"slug":"planning-irregular-object-packing-via","title":"Planning Irregular Object Packing via Hierarchical Reinforcement Learning","date":"2022-11-17","arxiv_id":"2211.09382","n_code_links":0,"syntology":null},{"paper":null,"slug":"solar-power-driven-ev-charging-optimization","title":"Solar Power driven EV Charging Optimization with Deep Reinforcement Learning","date":"2022-11-17","arxiv_id":"2211.09479","n_code_links":0,"syntology":null},{"paper":null,"slug":"addressing-the-issue-of-stochastic","title":"Addressing the issue of stochastic environments and local decision-making in multi-objective reinforcement learning","date":"2022-11-16","arxiv_id":"2211.08669","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-global-convergence-of-fitted-q","title":"On the Global Convergence of Fitted Q-Iteration with Two-layer Neural Network Parametrization","date":"2022-11-14","arxiv_id":"2211.07675","n_code_links":0,"syntology":null},{"paper":null,"slug":"nrem-and-rem-cognitive-and-energetic-effects","title":"NREM and REM: cognitive and energetic gains in thalamo-cortical sleeping and awake spiking model","date":"2022-11-13","arxiv_id":"2211.06889","n_code_links":0,"syntology":null},{"paper":"/paper/deep-w-networks-solving-multi-objective","slug":"deep-w-networks-solving-multi-objective","title":"Deep W-Networks: Solving Multi-Objective Optimisation Problems With Deep Reinforcement Learning","date":"2022-11-09","arxiv_id":"2211.04813","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforcement-learning-in-non-markovian","title":"Reinforcement Learning in Non-Markovian Environments","date":"2022-11-03","arxiv_id":"2211.01595","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-power-control","title":"Deep Reinforcement Learning for Power Control in Next-Generation WiFi Network Systems","date":"2022-11-02","arxiv_id":"2211.01107","n_code_links":0,"syntology":null},{"paper":"/paper/dynamiclight-dynamically-tuning-traffic","slug":"dynamiclight-dynamically-tuning-traffic","title":"DynamicLight: Two-Stage Dynamic Traffic Signal Timing","date":"2022-11-02","arxiv_id":"2211.01025","n_code_links":1,"syntology":null},{"paper":null,"slug":"offline-rl-with-realistic-datasets","title":"Offline RL With Realistic Datasets: Heteroskedasticity and Support Constraints","date":"2022-11-02","arxiv_id":"2211.01052","n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-learning-for-general-sum-low","title":"Representation Learning for General-sum Low-rank Markov Games","date":"2022-10-30","arxiv_id":"2210.16976","n_code_links":0,"syntology":null},{"paper":"/paper/defix-detecting-and-fixing-failure-scenarios","slug":"defix-detecting-and-fixing-failure-scenarios","title":"DeFIX: Detecting and Fixing Failure Scenarios with Reinforcement Learning in Imitation Learning Based Autonomous Driving","date":"2022-10-29","arxiv_id":"2210.16567","n_code_links":2,"syntology":null},{"paper":"/paper/solving-continuous-control-via-q-learning","slug":"solving-continuous-control-via-q-learning","title":"Solving Continuous Control via Q-learning","date":"2022-10-22","arxiv_id":"2210.12566","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["tseyde/decqn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sufficient-exploration-for-convex-q-learning","title":"Sufficient Exploration for Convex Q-learning","date":"2022-10-17","arxiv_id":"2210.09409","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-free-characterizations-of-the-hamilton","title":"Model-Free Characterizations of the Hamilton-Jacobi-Bellman Equation and Convex Q-Learning in Continuous Time","date":"2022-10-14","arxiv_id":"2210.08131","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-automatic-run","title":"Deep reinforcement learning for automatic run-time adaptation of UWB PHY radio settings","date":"2022-10-13","arxiv_id":"2210.15498","n_code_links":0,"syntology":null},{"paper":"/paper/hybrid-rl-using-both-offline-and-online-data","slug":"hybrid-rl-using-both-offline-and-online-data","title":"Hybrid RL: Using Both Offline and Online Data Can Make RL Efficient","date":"2022-10-13","arxiv_id":"2210.06718","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yudasong/hyq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/sustainable-online-reinforcement-learning-for","slug":"sustainable-online-reinforcement-learning-for","title":"Sustainable Online Reinforcement Learning for Auto-bidding","date":"2022-10-13","arxiv_id":"2210.07006","n_code_links":1,"syntology":null},{"paper":null,"slug":"censored-deep-reinforcement-patrolling-with","title":"Censored Deep Reinforcement Patrolling with Information Criterion for Monitoring Large Water Resources using Autonomous Surface Vehicles","date":"2022-10-12","arxiv_id":"2210.08115","n_code_links":0,"syntology":null},{"paper":"/paper/factors-of-influence-of-the-overestimation","slug":"factors-of-influence-of-the-overestimation","title":"Factors of Influence of the Overestimation Bias of Q-Learning","date":"2022-10-11","arxiv_id":"2210.05262","n_code_links":1,"syntology":null},{"paper":"/paper/pre-training-for-robots-offline-rl-enables","slug":"pre-training-for-robots-offline-rl-enables","title":"Pre-Training for Robots: Offline RL Enables Learning New Tasks from a Handful of Trials","date":"2022-10-11","arxiv_id":"2210.05178","n_code_links":1,"syntology":null},{"paper":null,"slug":"elastic-step-dqn-a-novel-multi-step-algorithm","title":"Elastic Step DQN: A novel multi-step algorithm to alleviate overestimation in Deep QNetworks","date":"2022-10-07","arxiv_id":"2210.03325","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-approach-for-multi","title":"Reinforcement Learning Approach for Multi-Agent Flexible Scheduling Problems","date":"2022-10-07","arxiv_id":"2210.03674","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-directed-controller-synthesis-via","slug":"scaling-directed-controller-synthesis-via","title":"Exploration Policies for On-the-Fly Controller Synthesis: A Reinforcement Learning Approach","date":"2022-10-07","arxiv_id":"2210.05393","n_code_links":1,"syntology":null},{"paper":"/paper/towards-safe-mechanical-ventilation-treatment","slug":"towards-safe-mechanical-ventilation-treatment","title":"Towards Safe Mechanical Ventilation Treatment Using Deep Offline Reinforcement Learning","date":"2022-10-05","arxiv_id":"2210.02552","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["FlemmingKondrup/DeepVent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"interpretable-option-discovery-using-deep-q","title":"Interpretable Option Discovery using Deep Q-Learning and Variational Autoencoders","date":"2022-10-03","arxiv_id":"2210.01231","n_code_links":0,"syntology":null},{"paper":null,"slug":"offline-reinforcement-learning-with-5","title":"Offline Reinforcement Learning with Differentiable Function Approximation is Provably Efficient","date":"2022-10-03","arxiv_id":"2210.00750","n_code_links":0,"syntology":null},{"paper":null,"slug":"bayesian-q-learning-with-imperfect-expert","title":"Bayesian Q-learning With Imperfect Expert Demonstrations","date":"2022-10-01","arxiv_id":"2210.01800","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-recurrent-q-learning-for-energy","title":"Deep Recurrent Q-learning for Energy-constrained Coverage with a Mobile Robot","date":"2022-10-01","arxiv_id":"2210.00327","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-convergence-of-average-reward-off-policy","title":"On Convergence of Average-Reward Off-Policy Control Algorithms in Weakly Communicating MDPs","date":"2022-09-30","arxiv_id":"2209.15141","n_code_links":0,"syntology":null},{"paper":null,"slug":"fire-a-failure-adaptive-reinforcement","title":"FIRE: A Failure-Adaptive Reinforcement Learning Framework for Edge Computing Migrations","date":"2022-09-28","arxiv_id":"2209.14399","n_code_links":0,"syntology":null},{"paper":null,"slug":"predictive-crypto-asset-automated-market","title":"Predictive Crypto-Asset Automated Market Making Architecture for Decentralized Finance using Deep Reinforcement Learning","date":"2022-09-28","arxiv_id":"2211.01346","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-hindsight-goal-relabeling","title":"Understanding Hindsight Goal Relabeling from a Divergence Minimization Perspective","date":"2022-09-26","arxiv_id":"2209.13046","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-discrete-soft-actor-critic","slug":"revisiting-discrete-soft-actor-critic","title":"Revisiting Discrete Soft Actor-Critic","date":"2022-09-21","arxiv_id":"2209.10081","n_code_links":1,"syntology":null},{"paper":null,"slug":"comparative-study-of-q-learning-and","title":"Comparative Study of Q-Learning and NeuroEvolution of Augmenting Topologies for Self Driving Agents","date":"2022-09-19","arxiv_id":"2209.09007","n_code_links":0,"syntology":null},{"paper":"/paper/man-multi-action-networks-learning","slug":"man-multi-action-networks-learning","title":"MAN: Multi-Action Networks Learning","date":"2022-09-19","arxiv_id":"2209.09329","n_code_links":1,"syntology":null},{"paper":null,"slug":"ma2ql-a-minimalist-approach-to-fully","title":"MA2QL: A Minimalist Approach to Fully Decentralized Multi-Agent Reinforcement Learning","date":"2022-09-17","arxiv_id":"2209.08244","n_code_links":0,"syntology":null},{"paper":"/paper/m-2-dqn-a-robust-method-for-accelerating-deep","slug":"m-2-dqn-a-robust-method-for-accelerating-deep","title":"M$^2$DQN: A Robust Method for Accelerating Deep Q-learning Network","date":"2022-09-16","arxiv_id":"2209.07809","n_code_links":1,"syntology":null},{"paper":"/paper/reducing-variance-in-temporal-difference","slug":"reducing-variance-in-temporal-difference","title":"Reducing Variance in Temporal-Difference Value Estimation via Ensemble of Deep Networks","date":"2022-09-16","arxiv_id":"2209.07670","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-cooperative-p2p","title":"Reinforcement Learning-Based Cooperative P2P Power Trading between DC Nanogrid Clusters with Wind and PV Energy Resources","date":"2022-09-16","arxiv_id":"2209.07744","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-task-1","title":"Deep Reinforcement Learning for Task Offloading in UAV-Aided Smart Farm Networks","date":"2022-09-15","arxiv_id":"2209.07367","n_code_links":0,"syntology":null},{"paper":null,"slug":"iot-aerial-base-station-task-offloading-with","title":"IoT-Aerial Base Station Task Offloading with Risk-Sensitive Reinforcement Learning for Smart Agriculture","date":"2022-09-15","arxiv_id":"2209.07382","n_code_links":0,"syntology":null},{"paper":null,"slug":"pathfinding-in-random-partially-observable","title":"Pathfinding in Random Partially Observable Environments with Vision-Informed Deep Reinforcement Learning","date":"2022-09-11","arxiv_id":"2209.04801","n_code_links":0,"syntology":null},{"paper":null,"slug":"structured-q-learning-for-antibody-design","title":"Structured Q-learning For Antibody Design","date":"2022-09-10","arxiv_id":"2209.04698","n_code_links":0,"syntology":null},{"paper":null,"slug":"continual-learning-benefits-from-multiple","title":"Continual learning benefits from multiple sleep mechanisms: NREM, REM, and Synaptic Downscaling","date":"2022-09-09","arxiv_id":"2209.05245","n_code_links":0,"syntology":null},{"paper":null,"slug":"double-q-learning-for-citizen-relocation","title":"Double Q-Learning for Citizen Relocation During Natural Hazards","date":"2022-09-08","arxiv_id":"2209.03800","n_code_links":0,"syntology":null},{"paper":"/paper/q-learning-decision-transformer-leveraging","slug":"q-learning-decision-transformer-leveraging","title":"Q-learning Decision Transformer: Leveraging Dynamic Programming for Conditional Sequence Modelling in Offline RL","date":"2022-09-08","arxiv_id":"2209.03993","n_code_links":1,"syntology":null},{"paper":"/paper/reward-delay-attacks-on-deep-reinforcement","slug":"reward-delay-attacks-on-deep-reinforcement","title":"Reward Delay Attacks on Deep Reinforcement Learning","date":"2022-09-08","arxiv_id":"2209.03540","n_code_links":1,"syntology":null},{"paper":null,"slug":"distilling-deep-rl-models-into-interpretable","title":"Distilling Deep RL Models Into Interpretable Neuro-Fuzzy Systems","date":"2022-09-07","arxiv_id":"2209.03357","n_code_links":0,"syntology":null},{"paper":null,"slug":"slatefree-a-model-free-decomposition-for","title":"SlateFree: a Model-Free Decomposition for Reinforcement Learning with Slate Actions","date":"2022-09-05","arxiv_id":"2209.01876","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-technique-to-create-weaker-abstract-board","title":"A Technique to Create Weaker Abstract Board Game Agents via Reinforcement Learning","date":"2022-09-01","arxiv_id":"2209.00711","n_code_links":0,"syntology":null}],"record_sha256":"84a03e733021529e56db64bdfcebe2bd6c6acaccce6c18d9853ca97ad51b6728","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}