{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/9","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":9,"pages_in_order":18,"rows_per_page":100,"rows":[801,900],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning/papers/8","next":"/method/q-learning/papers/10","papers":[{"paper":null,"slug":"multiple-correlated-jammers-nullification","title":"Multiple Correlated Jammers Nullification using LSTM-based Deep Dueling Neural Network","date":"2022-02-08","arxiv_id":"2202.03600","n_code_links":0,"syntology":null},{"paper":"/paper/skrl-modular-and-flexible-library-for","slug":"skrl-modular-and-flexible-library-for","title":"skrl: Modular and Flexible Library for Reinforcement Learning","date":"2022-02-08","arxiv_id":"2202.03825","n_code_links":1,"syntology":null},{"paper":"/paper/transfer-reinforcement-learning-for-differing","slug":"transfer-reinforcement-learning-for-differing","title":"Transfer Reinforcement Learning for Differing Action Spaces via Q-Network Representations","date":"2022-02-05","arxiv_id":"2202.02442","n_code_links":1,"syntology":null},{"paper":null,"slug":"generative-adversarial-exploration-for","title":"Generative Adversarial Exploration for Reinforcement Learning","date":"2022-01-27","arxiv_id":"2201.11685","n_code_links":0,"syntology":null},{"paper":"/paper/deep-q-learning-a-robust-control-approach","slug":"deep-q-learning-a-robust-control-approach","title":"Deep Q-learning: a robust control approach","date":"2022-01-21","arxiv_id":"2201.08610","n_code_links":1,"syntology":null},{"paper":null,"slug":"critic-algorithms-using-cooperative-networks","title":"Critic Algorithms using Cooperative Networks","date":"2022-01-19","arxiv_id":"2201.07839","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-improved-reinforcement-learning-algorithm","title":"An Improved Reinforcement Learning Algorithm for Learning to Branch","date":"2022-01-17","arxiv_id":"2201.06213","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-family-of-cognitively-realistic-parsing","title":"A Family of Cognitively Realistic Parsing Environments for Deep Reinforcement Learning","date":"2022-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"criticality-based-varying-step-number","title":"Criticality-Based Varying Step-Number Algorithm for Reinforcement Learning","date":"2022-01-13","arxiv_id":"2201.05034","n_code_links":0,"syntology":null},{"paper":null,"slug":"age-of-information-minimization-via","title":"Age-of-information minimization via opportunistic sampling by an energy harvesting source","date":"2022-01-08","arxiv_id":"2201.02787","n_code_links":0,"syntology":null},{"paper":null,"slug":"sales-time-series-analytics-using-deep-q","title":"Sales Time Series Analytics Using Deep Q-Learning","date":"2022-01-06","arxiv_id":"2201.02058","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-task","title":"Reinforcement Learning for Task Specifications with Action-Constraints","date":"2022-01-02","arxiv_id":"2201.00286","n_code_links":0,"syntology":null},{"paper":null,"slug":"operator-deep-q-learning-zero-shot-reward","title":"Operator Deep Q-Learning: Zero-Shot Reward Transferring in Reinforcement Learning","date":"2022-01-01","arxiv_id":"2201.00236","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-resolution-enhancement-plug-in-for","title":"A Resolution Enhancement Plug-in for Deformable Registration of Medical Images","date":"2021-12-30","arxiv_id":"2112.15180","n_code_links":0,"syntology":null},{"paper":"/paper/constraint-sampling-reinforcement-learning","slug":"constraint-sampling-reinforcement-learning","title":"Constraint Sampling Reinforcement Learning: Incorporating Expertise For Faster Learning","date":"2021-12-30","arxiv_id":"2112.15221","n_code_links":1,"syntology":null},{"paper":"/paper/polyak-ruppert-averaged-q-leaning-is","slug":"polyak-ruppert-averaged-q-leaning-is","title":"A Statistical Analysis of Polyak-Ruppert Averaged Q-learning","date":"2021-12-29","arxiv_id":"2112.14582","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lx10077/AveQLearning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-graph-attention-learning-approach-to","title":"A Graph Attention Learning Approach to Antenna Tilt Optimization","date":"2021-12-27","arxiv_id":"2112.14843","n_code_links":0,"syntology":null},{"paper":"/paper/intelligent-traffic-light-via-policy-based","slug":"intelligent-traffic-light-via-policy-based","title":"Intelligent Traffic Light via Policy-based Deep Reinforcement Learning","date":"2021-12-27","arxiv_id":"2112.13817","n_code_links":1,"syntology":null},{"paper":"/paper/lane-change-decision-making-through-deep-1","slug":"lane-change-decision-making-through-deep-1","title":"Lane Change Decision-Making through Deep Reinforcement Learning","date":"2021-12-24","arxiv_id":"2112.14705","n_code_links":2,"syntology":null},{"paper":null,"slug":"local-advantage-networks-for-cooperative","title":"Local Advantage Networks for Cooperative Multi-Agent Reinforcement Learning","date":"2021-12-23","arxiv_id":"2112.12458","n_code_links":0,"syntology":null},{"paper":"/paper/safety-and-liveness-guarantees-through-reach","slug":"safety-and-liveness-guarantees-through-reach","title":"Safety and Liveness Guarantees through Reach-Avoid Reinforcement Learning","date":"2021-12-23","arxiv_id":"2112.12288","n_code_links":1,"syntology":null},{"paper":null,"slug":"aerial-base-station-positioning-and-power","title":"Aerial Base Station Positioning and Power Control for Securing Communications: A Deep Q-Network Approach","date":"2021-12-21","arxiv_id":"2112.11090","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deep-reinforcement-learning-model-for","title":"A deep reinforcement learning model for predictive maintenance planning of road assets: Integrating LCA and LCCA","date":"2021-12-20","arxiv_id":"2112.12589","n_code_links":0,"syntology":null},{"paper":"/paper/space-non-cooperative-object-active-tracking","slug":"space-non-cooperative-object-active-tracking","title":"Space Non-cooperative Object Active Tracking with Deep Reinforcement Learning","date":"2021-12-18","arxiv_id":"2112.09854","n_code_links":1,"syntology":null},{"paper":null,"slug":"finite-sample-analysis-of-decentralized-q","title":"Finite-Sample Analysis of Decentralized Q-Learning for Stochastic Games","date":"2021-12-15","arxiv_id":"2112.07859","n_code_links":0,"syntology":null},{"paper":null,"slug":"scientific-discovery-and-the-cost-of","title":"Scientific Discovery and the Cost of Measurement -- Balancing Information and Cost in Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07535","n_code_links":0,"syntology":null},{"paper":null,"slug":"teaching-a-robot-to-walk-using-reinforcement","title":"Teaching a Robot to Walk Using Reinforcement Learning","date":"2021-12-13","arxiv_id":"2112.07031","n_code_links":0,"syntology":null},{"paper":null,"slug":"control-tutored-reinforcement-learning-1","title":"Control-Tutored Reinforcement Learning: Towards the Integration of Data-Driven and Model-Based Control","date":"2021-12-11","arxiv_id":"2112.06018","n_code_links":0,"syntology":null},{"paper":"/paper/deep-q-network-with-proximal-iteration-1","slug":"deep-q-network-with-proximal-iteration-1","title":"Faster Deep Reinforcement Learning with Slower Online Network","date":"2021-12-10","arxiv_id":"2112.05848","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["amazon-research/fast-rl-with-slow-updates"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"quantum-architecture-search-via-continual","title":"Quantum Architecture Search via Continual Reinforcement Learning","date":"2021-12-10","arxiv_id":"2112.05779","n_code_links":0,"syntology":null},{"paper":null,"slug":"application-of-deep-reinforcement-learning-to","title":"Application of Deep Reinforcement Learning to Payment Fraud","date":"2021-12-08","arxiv_id":"2112.04236","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-results-for-q-learning-with","title":"Convergence Results For Q-Learning With Experience Replay","date":"2021-12-08","arxiv_id":"2112.04213","n_code_links":0,"syntology":null},{"paper":null,"slug":"replay-for-safety","title":"Replay For Safety","date":"2021-12-08","arxiv_id":"2112.04229","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-risk-averse-preview-based-q-learning","title":"A Risk-Averse Preview-based $Q$-Learning Algorithm: Application to Highway Driving of Autonomous Vehicles","date":"2021-12-06","arxiv_id":"2112.03232","n_code_links":0,"syntology":null},{"paper":null,"slug":"faster-non-asymptotic-convergence-for-double","title":"Faster Non-asymptotic Convergence for Double Q-learning","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/solving-reward-collecting-problems-with-uavs","slug":"solving-reward-collecting-problems-with-uavs","title":"Solving reward-collecting problems with UAVs: a comparison of online optimization and Q-learning","date":"2021-11-30","arxiv_id":"2112.00141","n_code_links":1,"syntology":null},{"paper":null,"slug":"deepcq-robust-and-scalable-routing-with-multi","title":"DeepCQ+: Robust and Scalable Routing with Multi-Agent Deep Reinforcement Learning for Highly Dynamic Networks","date":"2021-11-29","arxiv_id":"2111.15013","n_code_links":0,"syntology":null},{"paper":null,"slug":"final-adaptation-reinforcement-learning-for-n","title":"Final Adaptation Reinforcement Learning for N-Player Games","date":"2021-11-29","arxiv_id":"2111.14375","n_code_links":0,"syntology":null},{"paper":null,"slug":"count-based-temperature-scheduling-for","title":"Count-Based Temperature Scheduling for Maximum Entropy Reinforcement Learning","date":"2021-11-28","arxiv_id":"2111.14204","n_code_links":0,"syntology":null},{"paper":"/paper/deep-q-learning-based-reinforcement-learning","slug":"deep-q-learning-based-reinforcement-learning","title":"Deep Q-Learning based Reinforcement Learning Approach for Network Intrusion Detection","date":"2021-11-27","arxiv_id":"2111.13978","n_code_links":1,"syntology":null},{"paper":"/paper/gdi-rethinking-what-makes-reinforcement-1","slug":"gdi-rethinking-what-makes-reinforcement-1","title":"GDI: Rethinking What Makes Reinforcement Learning Different from Supervised Learning","date":"2021-11-24","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"multicrew-scheduling-and-routing-in-road","title":"Multicrew Scheduling and Routing in Road Network Restoration Based on Deep Q-learning","date":"2021-11-24","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"reversible-action-design-for-combinatorial-1","title":"Reversible Action Design for Combinatorial Optimization with ReinforcementLearning","date":"2021-11-24","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/understanding-the-impact-of-data-distribution","slug":"understanding-the-impact-of-data-distribution","title":"The Impact of Data Distribution on Q-learning with Function Approximation","date":"2021-11-23","arxiv_id":"2111.11758","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-agent-bayesian-deep-reinforcement","title":"Multi-agent Bayesian Deep Reinforcement Learning for Microgrid Energy Management under Communication Failures","date":"2021-11-22","arxiv_id":"2111.11868","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-improved-reinforcement-learning-model","title":"An Improved Reinforcement Learning Model Based on Sentiment Analysis","date":"2021-11-19","arxiv_id":"2111.15354","n_code_links":0,"syntology":null},{"paper":null,"slug":"improved-method-of-stock-trading-under","title":"Improved Method of Stock Trading under Reinforcement Learning Based on DRQN and Sentiment Indicators ARBR","date":"2021-11-19","arxiv_id":"2111.15356","n_code_links":0,"syntology":null},{"paper":null,"slug":"aggressive-q-learning-with-ensembles-1","title":"Aggressive Q-Learning with Ensembles: Achieving Both High Sample Efficiency and High Asymptotic Performance","date":"2021-11-17","arxiv_id":"2111.09159","n_code_links":0,"syntology":null},{"paper":null,"slug":"consecutive-task-oriented-dialog-policy","title":"Consecutive Task-oriented Dialog Policy Learning","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"where-to-look-a-unified-attention-model-for","title":"Where to Look: A Unified Attention Model for Visual Recognition with Reinforcement Learning","date":"2021-11-13","arxiv_id":"2111.07169","n_code_links":0,"syntology":null},{"paper":"/paper/improving-experience-replay-through-modeling","slug":"improving-experience-replay-through-modeling","title":"Improving Experience Replay through Modeling of Similar Transitions' Sets","date":"2021-11-12","arxiv_id":"2111.06907","n_code_links":1,"syntology":null},{"paper":null,"slug":"q-learning-for-mdps-with-general-spaces","title":"Q-Learning for MDPs with General Spaces: Convergence and Near Optimality via Quantization under Weak Continuity","date":"2021-11-12","arxiv_id":"2111.06781","n_code_links":0,"syntology":null},{"paper":"/paper/good-robot-now-watch-this-repurposing","slug":"good-robot-now-watch-this-repurposing","title":"\"Good Robot! Now Watch This!\": Repurposing Reinforcement Learning for Task-to-Task Transfer","date":"2021-11-08","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/guiding-multi-step-rearrangement-tasks-with","slug":"guiding-multi-step-rearrangement-tasks-with","title":"Guiding Multi-Step Rearrangement Tasks with Natural Language Instructions","date":"2021-11-08","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"on-assessing-the-safety-of-reinforcement","title":"On Assessing The Safety of Reinforcement Learning algorithms Using Formal Methods","date":"2021-11-08","arxiv_id":"2111.04865","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-rna-secondary-structure-design","title":"Improving RNA Secondary Structure Design using Deep Reinforcement Learning","date":"2021-11-05","arxiv_id":"2111.04504","n_code_links":0,"syntology":null},{"paper":null,"slug":"supervised-advantage-actor-critic-for","title":"Supervised Advantage Actor-Critic for Recommender Systems","date":"2021-11-05","arxiv_id":"2111.03474","n_code_links":0,"syntology":null},{"paper":null,"slug":"balanced-q-learning-combining-the-influence","title":"Balanced Q-learning: Combining the Influence of Optimistic and Pessimistic Targets","date":"2021-11-03","arxiv_id":"2111.02787","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-service-provisioning-in-nfv-enabled","title":"Online Service Provisioning in NFV-enabled Networks Using Deep Reinforcement Learning","date":"2021-11-03","arxiv_id":"2111.02209","n_code_links":0,"syntology":null},{"paper":"/paper/koopman-q-learning-offline-reinforcement-1","slug":"koopman-q-learning-offline-reinforcement-1","title":"Koopman Q-learning: Offline Reinforcement Learning via Symmetries of Dynamics","date":"2021-11-02","arxiv_id":"2111.01365","n_code_links":0,"syntology":null},{"paper":"/paper/human-level-control-without-server-grade-1","slug":"human-level-control-without-server-grade-1","title":"Human-Level Control without Server-Grade Hardware","date":"2021-11-01","arxiv_id":"2111.01264","n_code_links":1,"syntology":null},{"paper":null,"slug":"decentralized-multi-agent-reinforcement-3","title":"Decentralized Multi-Agent Reinforcement Learning: An Off-Policy Method","date":"2021-10-31","arxiv_id":"2111.00438","n_code_links":0,"syntology":null},{"paper":null,"slug":"throughput-and-latency-in-the-distributed-q","title":"Throughput and Latency in the Distributed Q-Learning Random Access mMTC Networks","date":"2021-10-30","arxiv_id":"2111.00299","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-communicate-with-reinforcement","title":"Learning to Communicate with Reinforcement Learning for an Adaptive Traffic Control System","date":"2021-10-29","arxiv_id":"2110.15779","n_code_links":0,"syntology":null},{"paper":null,"slug":"location-routing-optimisation-for-urban","title":"Location-routing Optimisation for Urban Logistics Using Mobile Parcel Locker Based on Hybrid Q-Learning Algorithm","date":"2021-10-29","arxiv_id":"2110.15485","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-aided-packet","title":"Deep Reinforcement Learning Aided Packet-Routing For Aeronautical Ad-Hoc Networks Formed by Passenger Planes","date":"2021-10-28","arxiv_id":"2110.15146","n_code_links":0,"syntology":null},{"paper":null,"slug":"temporal-difference-value-estimation-via","title":"Temporal-Difference Value Estimation via Uncertainty-Guided Soft Updates","date":"2021-10-28","arxiv_id":"2110.14818","n_code_links":0,"syntology":null},{"paper":null,"slug":"finite-horizon-q-learning-stability","title":"Finite Horizon Q-learning: Stability, Convergence, Simulations and an application on Smart Grids","date":"2021-10-27","arxiv_id":"2110.15093","n_code_links":0,"syntology":null},{"paper":null,"slug":"v-learning-a-simple-efficient-decentralized","title":"V-Learning -- A Simple, Efficient, Decentralized Algorithm for Multiagent RL","date":"2021-10-27","arxiv_id":"2110.14555","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-dpdk-based-acceleration-method-for","title":"Accelerating Distributed Deep Reinforcement Learning by In-Network Experience Sampling","date":"2021-10-26","arxiv_id":"2110.13506","n_code_links":0,"syntology":null},{"paper":"/paper/distributional-reinforcement-learning-for-4","slug":"distributional-reinforcement-learning-for-4","title":"Distributional Reinforcement Learning for Multi-Dimensional Reward Functions","date":"2021-10-26","arxiv_id":"2110.13578","n_code_links":2,"syntology":null},{"paper":"/paper/multi-agent-advisor-q-learning","slug":"multi-agent-advisor-q-learning","title":"Multi-Agent Advisor Q-Learning","date":"2021-10-26","arxiv_id":"2111.00345","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-q-learning-solve-multi-armed-bantids","title":"Can Q-learning solve Multi Armed Bantids?","date":"2021-10-21","arxiv_id":"2110.10934","n_code_links":0,"syntology":null},{"paper":null,"slug":"more-efficient-exploration-with-symbolic","title":"More Efficient Exploration with Symbolic Priors on Action Sequence Equivalences","date":"2021-10-20","arxiv_id":"2110.10632","n_code_links":0,"syntology":null},{"paper":"/paper/playing-2048-with-reinforcement-learning","slug":"playing-2048-with-reinforcement-learning","title":"Playing 2048 With Reinforcement Learning","date":"2021-10-20","arxiv_id":"2110.10374","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-q-learning-based-approach-for-distributed","title":"A Q-Learning-based Approach for Distributed Beam Scheduling in mmWave Networks","date":"2021-10-17","arxiv_id":"2110.08704","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-target-q-learning-with-reverse-1","title":"Online Target Q-learning with Reverse Experience Replay: Efficiently finding the Optimal Policy for Linear MDPs","date":"2021-10-16","arxiv_id":"2110.08440","n_code_links":0,"syntology":null},{"paper":null,"slug":"value-penalized-q-learning-for-recommender","title":"Value Penalized Q-Learning for Recommender Systems","date":"2021-10-15","arxiv_id":"2110.07923","n_code_links":0,"syntology":null},{"paper":null,"slug":"decentralized-cooperative-multi-agent-1","title":"On Improving Model-Free Algorithms for Decentralized Multi-Agent Reinforcement Learning","date":"2021-10-12","arxiv_id":"2110.05707","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-block-linear-system-solver-using-q","title":"Fast Block Linear System Solver Using Q-Learning Schduling for Unified Dynamic Power System Simulations","date":"2021-10-12","arxiv_id":"2110.05843","n_code_links":0,"syntology":null},{"paper":"/paper/bridging-the-gap-between-label-and-reference-1","slug":"bridging-the-gap-between-label-and-reference-1","title":"Bridging the Gap between Label- and Reference-based Synthesis in Multi-attribute Image-to-Image Translation","date":"2021-10-11","arxiv_id":"2110.05055","n_code_links":1,"syntology":null},{"paper":null,"slug":"navigation-in-urban-environments-amongst","title":"Navigation In Urban Environments Amongst Pedestrians Using Multi-Objective Deep Reinforcement Learning","date":"2021-10-11","arxiv_id":"2110.05205","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deep-learning-inference-scheme-based-on","title":"A Deep Learning Inference Scheme Based on Pipelined Matrix Multiplication Acceleration Design and Non-uniform Quantization","date":"2021-10-10","arxiv_id":"2110.04861","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-in-two-player-zero-sum","slug":"reinforcement-learning-in-two-player-zero-sum","title":"Reinforcement Learning In Two Player Zero Sum Simultaneous Action Games","date":"2021-10-10","arxiv_id":"2110.04835","n_code_links":1,"syntology":null},{"paper":null,"slug":"breaking-the-sample-complexity-barrier-to","title":"Breaking the Sample Complexity Barrier to Regret-Optimal Model-Free Reinforcement Learning","date":"2021-10-09","arxiv_id":"2110.04645","n_code_links":0,"syntology":null},{"paper":"/paper/training-transition-policies-via-distribution-1","slug":"training-transition-policies-via-distribution-1","title":"Training Transition Policies via Distribution Matching for Complex Tasks","date":"2021-10-08","arxiv_id":"2110.04357","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-guidewire","title":"Deep reinforcement learning for guidewire navigation in coronary artery phantom","date":"2021-10-05","arxiv_id":"2110.01840","n_code_links":0,"syntology":null},{"paper":"/paper/dropout-q-functions-for-doubly-efficient","slug":"dropout-q-functions-for-doubly-efficient","title":"Dropout Q-Functions for Doubly Efficient Reinforcement Learning","date":"2021-10-05","arxiv_id":"2110.02034","n_code_links":2,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["TakuyaHiraoka/Dropout-Q-Functions-for-Doubly-Efficient-Reinforcement-Learning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/uncertainty-based-offline-reinforcement","slug":"uncertainty-based-offline-reinforcement","title":"Uncertainty-Based Offline Reinforcement Learning with Diversified Q-Ensemble","date":"2021-10-04","arxiv_id":"2110.01548","n_code_links":5,"syntology":{"ran":13,"of":21,"n_ran_checked":12,"n_instrument":1,"unverified":8,"pointer_only":6,"phrase":"13 ran (of which 11 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","official":null}},{"paper":null,"slug":"parallel-actors-and-learners-a-framework-for","title":"Parallel Actors and Learners: A Framework for Generating Scalable RL Implementations","date":"2021-10-03","arxiv_id":"2110.01101","n_code_links":0,"syntology":null},{"paper":null,"slug":"cellular-traffic-offloading-via-opportunistic","title":"Cellular traffic offloading via Opportunistic Networking with Reinforcement Learning","date":"2021-10-01","arxiv_id":"2110.00397","n_code_links":0,"syntology":null},{"paper":null,"slug":"motion-planning-for-autonomous-vehicles-in","title":"Motion Planning for Autonomous Vehicles in the Presence of Uncertainty Using Reinforcement Learning","date":"2021-10-01","arxiv_id":"2110.00640","n_code_links":0,"syntology":null},{"paper":"/paper/learning-the-markov-decision-process-in-the","slug":"learning-the-markov-decision-process-in-the","title":"Learning the Markov Decision Process in the Sparse Gaussian Elimination","date":"2021-09-30","arxiv_id":"2109.14929","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-q-learning-for-interaction-limited","title":"Adaptive Q-learning for Interaction-Limited Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"an-attempt-to-model-human-trust-with","title":"An Attempt to Model Human Trust with Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"better-state-exploration-using-action","title":"Better state exploration using action sequence equivalence","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"bootstrapped-hindsight-experience-replay-with","title":"Bootstrapped Hindsight Experience replay with Counterintuitive Prioritization","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"convergent-and-efficient-deep-q-learning","title":"Convergent and Efficient Deep Q Learning Algorithm","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"decentralized-cooperative-multi-agent","title":"Decentralized Cooperative Multi-Agent Reinforcement Learning with Exploration","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-q-learning-for-intelligent","slug":"deep-reinforcement-q-learning-for-intelligent","title":"Deep Reinforcement Q-Learning for Intelligent Traffic Signal Control with Partial Detection","date":"2021-09-29","arxiv_id":"2109.14337","n_code_links":1,"syntology":null}],"record_sha256":"917024920820fbbce7cf2c06704f7b2e39745ac6ab10b1545b32732b7f3e9006","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}