{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/q-learning/papers/16","list_of":"/task/q-learning","task":"Q-Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":16,"pages_in_order":20,"rows_per_page":100,"rows":[1501,1600],"of":1918,"counts":{"archive_papers_tagged":1918,"with_a_code_link":463,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1918,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":102,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":102,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/q-learning","prev":"/task/q-learning/papers/15","next":"/task/q-learning/papers/17","papers":[{"url":null,"slug":"should-artificial-agents-ask-for-help-in","title":"Should artificial agents ask for help in human-robot collaborative problem-solving?","date":"2020-05-25","arxiv_id":"2006.00882","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-decision","title":"A reinforcement learning based decision support system in textile manufacturing process","date":"2020-05-20","arxiv_id":"2005.09867","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-learning-for-near-optimal-scheduling","title":"Safe Learning for Near Optimal Scheduling","date":"2020-05-19","arxiv_id":"2005.09253","repositories_listed":0,"syntology":null},{"url":null,"slug":"basal-glucose-control-in-type-1-diabetes","title":"Basal Glucose Control in Type 1 Diabetes using Deep Reinforcement Learning: An In Silico Validation","date":"2020-05-18","arxiv_id":"2005.09059","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-augmented-entropy-regularized","title":"Entropy-Augmented Entropy-Regularized Reinforcement Learning and a Continuous Path from Policy Gradient to Q-Learning","date":"2020-05-18","arxiv_id":"2005.08844","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-q-learning-genetic-algorithms-based","title":"A Deep Q-learning/genetic Algorithms Based Novel Methodology For Optimizing Covid-19 Pandemic Government Actions","date":"2020-05-15","arxiv_id":"2005.07656","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-to-2","title":"A Deep Reinforcement Learning Approach to Efficient Drone Mobility Support","date":"2020-05-11","arxiv_id":"2005.05229","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-fpga-based-on-device-reinforcement","title":"An FPGA-Based On-Device Reinforcement Learning Approach using Online Sequential Learning","date":"2020-05-10","arxiv_id":"2005.04646","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-thermostatically","title":"Reinforcement Learning for Thermostatically Controlled Loads Control using Modelica and Python","date":"2020-05-09","arxiv_id":"2005.04444","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-beam-association-for-high-mobility","title":"Optimal Beam Association for High Mobility mmWave Vehicular Networks: Lightweight Parallel Reinforcement Learning Approach","date":"2020-05-02","arxiv_id":"2005.00694","repositories_listed":0,"syntology":null},{"url":null,"slug":"implementing-inductive-bias-for-different","title":"Implementing Inductive bias for different navigation tasks through diverse RNN attrractors","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-efficient-parameter-server","title":"Learning Efficient Parameter Server Synchronization Policies for Distributed SGD","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"whittle-index-based-q-learning-for-restless","title":"Whittle index based Q-learning for restless bandits with average reward","date":"2020-04-29","arxiv_id":"2004.14427","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolution-of-q-values-for-deep-q-learning-in","title":"Evolution of Q Values for Deep Q Learning in Stable Baselines","date":"2020-04-24","arxiv_id":"2004.11766","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dialog-policies-from-weak","title":"Learning Dialog Policies from Weak Demonstrations","date":"2020-04-23","arxiv_id":"2004.11054","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficient-power-allocation-and-q","title":"Energy-Efficient Power Allocation and Q-Learning-Based Relay Selection for Relay-Aided D2D Communication","date":"2020-04-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-querying-for-target-tracking-in","title":"Intelligent Querying for Target Tracking in Camera Networks using Deep Q-Learning with n-Step Bootstrapping","date":"2020-04-20","arxiv_id":"2004.09632","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-adaptive-1","title":"Deep Reinforcement Learning for Adaptive Learning Systems","date":"2020-04-17","arxiv_id":"2004.08410","repositories_listed":0,"syntology":null},{"url":null,"slug":"show-us-the-way-learning-to-manage-dialog","title":"Show Us the Way: Learning to Manage Dialog from Demonstrations","date":"2020-04-17","arxiv_id":"2004.08114","repositories_listed":0,"syntology":null},{"url":null,"slug":"k-spin-hamiltonian-for-quantum-resolvable","title":"K-spin Hamiltonian for quantum-resolvable Markov decision processes","date":"2020-04-13","arxiv_id":"2004.06040","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-learning-of-text-adventure-games","title":"Zero-Shot Learning of Text Adventure Games with Sentence-Level Semantics","date":"2020-04-06","arxiv_id":"2004.02986","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-3","title":"Multi-agent Reinforcement Learning for Resource Allocation in IoT networks with Edge Computing","date":"2020-04-05","arxiv_id":"2004.02315","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimizing-age-of-information-for-fog","title":"Minimizing Age-of-Information for Fog Computing-supported Vehicular Networks with Deep Q-learning","date":"2020-04-04","arxiv_id":"2004.04640","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-mixed-integer","title":"Reinforcement Learning for Mixed-Integer Problems Based on MPC","date":"2020-04-03","arxiv_id":"2004.01430","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-via-projection-on","title":"Safe Reinforcement Learning via Projection on a Safe Set: How to Achieve Optimality?","date":"2020-04-02","arxiv_id":"2004.00915","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistically-model-checking-pctl","title":"Statistically Model Checking PCTL Specifications on Markov Decision Processes via Reinforcement Learning","date":"2020-04-01","arxiv_id":"2004.00273","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-rolling-horizon-evolution-algorithm","title":"Enhanced Rolling Horizon Evolution Algorithm with Opponent Model Learning: Results for the Fighting Game AI Competition","date":"2020-03-31","arxiv_id":"2003.13949","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-medical-triage-from-clinicians-using","title":"Learning medical triage from clinicians using Deep Q-Learning","date":"2020-03-28","arxiv_id":"2003.12828","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-distributional-analysis-of-sampling-based","title":"A Distributional Analysis of Sampling-Based Reinforcement Learning Algorithms","date":"2020-03-27","arxiv_id":"2003.12239","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-q-learning","title":"Robust Q-learning","date":"2020-03-27","arxiv_id":"2003.12427","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-of-recursive-stochastic","title":"Convergence of Recursive Stochastic Algorithms using Wasserstein Divergence","date":"2020-03-25","arxiv_id":"2003.11403","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-in-regularized-mean-field-games","title":"Q-Learning in Regularized Mean-field Games","date":"2020-03-24","arxiv_id":"2003.12151","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-reinforcement-learning-for-1","title":"Distributed Reinforcement Learning for Cooperative Multi-Robot Object Manipulation","date":"2020-03-21","arxiv_id":"2003.09540","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-weighted-q","title":"Deep Reinforcement Learning with Weighted Q-Learning","date":"2020-03-20","arxiv_id":"2003.09280","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-multi-time-scale-constraints-in","title":"Deep Constrained Q-learning","date":"2020-03-20","arxiv_id":"2003.09398","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-perception-and-representation-for","title":"Active Perception and Representation for Robotic Manipulation","date":"2020-03-15","arxiv_id":"2003.06734","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-general-framework-for-learning-mean-field","title":"A General Framework for Learning Mean-Field Games","date":"2020-03-13","arxiv_id":"2003.06069","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-of-deep-q-network-in-portfolio","title":"Application of Deep Q-Network in Portfolio Management","date":"2020-03-13","arxiv_id":"2003.06365","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-algorithm-and-regret-analysis-for","title":"Provably Efficient Model-Free Algorithm for MDPs with Peak Constraints","date":"2020-03-11","arxiv_id":"2003.05555","repositories_listed":0,"syntology":null},{"url":null,"slug":"indirect-and-direct-training-of-spiking","title":"Indirect and Direct Training of Spiking Neural Networks for End-to-End Control of a Lane-Keeping Vehicle","date":"2020-03-10","arxiv_id":"2003.04603","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-cost-management-in-smart-meters-using","title":"Privacy-Cost Management in Smart Meters Using Deep Reinforcement Learning","date":"2020-03-10","arxiv_id":"2003.04946","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-planning-for-connected-autonomous","title":"A Multi-Agent Reinforcement Learning Approach For Safe and Efficient Behavior Planning Of Connected Autonomous Vehicles","date":"2020-03-09","arxiv_id":"2003.04371","repositories_listed":0,"syntology":null},{"url":null,"slug":"software-level-accuracy-using-stochastic","title":"Software-Level Accuracy Using Stochastic Computing With Charge-Trap-Flash Based Weight Matrix","date":"2020-03-09","arxiv_id":"2004.11120","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-reinforcement-learning-under","title":"Transfer Reinforcement Learning under Unobserved Contextual Information","date":"2020-03-09","arxiv_id":"2003.04427","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-cooperative","title":"Reinforcement Learning Based Cooperative Coded Caching under Dynamic Popularities in Ultra-Dense Networks","date":"2020-03-08","arxiv_id":"2003.03758","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-object-level-deep","title":"Relevance-Guided Modeling of Object Dynamics for Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01384","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-structural-hyper-parameter","title":"Adaptive Structural Hyper-Parameter Configuration by Q-Learning","date":"2020-03-02","arxiv_id":"2003.00863","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-policy-reuse-using-deep-mixture","title":"Contextual Policy Transfer in Reinforcement Learning Domains via Deep Mixtures-of-Experts","date":"2020-02-29","arxiv_id":"2003.00203","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-flipit","title":"Deep Reinforcement Learning for FlipIt Security Game","date":"2020-02-28","arxiv_id":"2002.12909","repositories_listed":0,"syntology":null},{"url":null,"slug":"g-learner-and-girl-goal-based-wealth","title":"G-Learner and GIRL: Goal Based Wealth Management with Reinforcement Learning","date":"2020-02-25","arxiv_id":"2002.10990","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneously-evolving-deep-reinforcement","title":"Simultaneously Evolving Deep Reinforcement Learning Models using Multifactorial Optimization","date":"2020-02-25","arxiv_id":"2002.12133","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-double-q-learning-approach-for-navigation","title":"A Double Q-Learning Approach for Navigation of Aerial Vehicles with Connectivity Constraint","date":"2020-02-24","arxiv_id":"2002.10563","repositories_listed":0,"syntology":null},{"url":null,"slug":"millimeter-wave-communications-with-an","title":"Millimeter Wave Communications with an Intelligent Reflector: Performance Optimization and Distributional Reinforcement Learning","date":"2020-02-24","arxiv_id":"2002.10572","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-with-uniformly-bounded-variance","title":"Q-learning with Uniformly Bounded Variance: Large Discounting is Not a Barrier to Fast Learning","date":"2020-02-24","arxiv_id":"2002.10301","repositories_listed":0,"syntology":null},{"url":null,"slug":"periodic-q-learning","title":"Periodic Q-Learning","date":"2020-02-23","arxiv_id":"2002.09795","repositories_listed":0,"syntology":null},{"url":null,"slug":"anypath-routing-protocol-design-via-q","title":"Anypath Routing Protocol Design via Q-Learning for Underwater Sensor Networks","date":"2020-02-22","arxiv_id":"2002.09623","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-aided-search-and-rescue-operation-using","title":"UAV Aided Search and Rescue Operation Using Reinforcement Learning","date":"2020-02-19","arxiv_id":"2002.08415","repositories_listed":0,"syntology":null},{"url":null,"slug":"agnostic-q-learning-with-function","title":"Agnostic Q-learning with Function Approximation in Deterministic Systems: Tight Bounds on Approximation Error and Sample Complexity","date":"2020-02-17","arxiv_id":"2002.07125","repositories_listed":0,"syntology":null},{"url":null,"slug":"listwise-learning-to-rank-with-deep-q","title":"Listwise Learning to Rank with Deep Q-Networks","date":"2020-02-13","arxiv_id":"2002.07651","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-bounds-for-discounted-mdps","title":"Regret Bounds for Discounted MDPs","date":"2020-02-12","arxiv_id":"2002.05138","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-for-mean-field-controls","title":"Mean-Field Controls with Q-learning for Cooperative MARL: Convergence and Complexity Analysis","date":"2020-02-10","arxiv_id":"2002.04131","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-detection-of-maximum-common-subgraph-via","title":"GLSearch: Maximum Common Subgraph Detection via Learning to Search","date":"2020-02-08","arxiv_id":"2002.03129","repositories_listed":0,"syntology":null},{"url":null,"slug":"manipulating-reinforcement-learning-poisoning","title":"Manipulating Reinforcement Learning: Poisoning Attacks on Cost Signals","date":"2020-02-07","arxiv_id":"2002.03827","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-wasserstein-constrained-deep-q-learning","title":"Safe Wasserstein Constrained Deep Q-Learning","date":"2020-02-07","arxiv_id":"2002.03016","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-sample-analysis-of-stochastic","title":"Finite-Sample Analysis of Stochastic Approximation Using Smooth Convex Envelopes","date":"2020-02-03","arxiv_id":"2002.00874","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-analysis-of-asynchronous","title":"Finite-Time Analysis of Asynchronous Stochastic Approximation and $Q$-Learning","date":"2020-02-01","arxiv_id":"2002.00260","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-control-of-a-line-follower-robot","title":"Autonomous Control of a Line Follower Robot Using a Q-Learning Controller","date":"2020-01-23","arxiv_id":"2001.08841","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-in-enormous-action-spaces-via","title":"Q-Learning in enormous action spaces via amortized approximate maximization","date":"2020-01-22","arxiv_id":"2001.08116","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-multi-agent-reinforcement","title":"Model-based Multi-Agent Reinforcement Learning with Cooperative Prioritized Sweeping","date":"2020-01-15","arxiv_id":"2001.07527","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-sequential-resource-investment-planning","title":"A storage expansion planning framework using reinforcement learning and simulation-based optimization","date":"2020-01-10","arxiv_id":"2001.03507","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-probabilistic-simulator-of-spatial-demand","title":"A Probabilistic Simulator of Spatial Demand for Product Allocation","date":"2020-01-09","arxiv_id":"2001.03210","repositories_listed":0,"syntology":null},{"url":null,"slug":"eeg-based-drowsiness-estimation-for-driving","title":"EEG-based Drowsiness Estimation for Driving Safety using Deep Q-Learning","date":"2020-01-08","arxiv_id":"2001.02399","repositories_listed":0,"syntology":null},{"url":null,"slug":"experimental-analysis-of-reinforcement","title":"Experimental Analysis of Reinforcement Learning Techniques for Spectrum Sharing Radar","date":"2020-01-06","arxiv_id":"2001.01799","repositories_listed":0,"syntology":null},{"url":null,"slug":"svqn-sequential-variational-soft-q-learning","title":"SVQN: Sequential Variational Soft Q-Learning Networks","date":"2020-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"way-off-policy-batch-deep-reinforcement-1","title":"Way Off-Policy Batch Deep Reinforcement Learning of Human Preferences in Dialog","date":"2020-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fitted-q-learning-in-mean-field-games","title":"Learning in Discounted-cost and Average-cost Mean-field Games","date":"2019-12-31","arxiv_id":"1912.13309","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-theoretic-model-predictive-q-1","title":"Information Theoretic Model Predictive Q-Learning","date":"2019-12-31","arxiv_id":"2001.02153","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-gamblers-problem-and-beyond-1","title":"The Gambler's Problem and Beyond","date":"2019-12-31","arxiv_id":"2001.00102","repositories_listed":0,"syntology":null},{"url":null,"slug":"hamilton-jacobi-bellman-equations-for-q","title":"Hamilton-Jacobi-Bellman Equations for Q-Learning in Continuous Time","date":"2019-12-23","arxiv_id":"1912.10697","repositories_listed":0,"syntology":null},{"url":null,"slug":"soft-q-network","title":"Soft Q Network","date":"2019-12-20","arxiv_id":"1912.10891","repositories_listed":0,"syntology":null},{"url":null,"slug":"sepsis-world-model-a-mimic-based-openai-gym","title":"Sepsis World Model: A MIMIC-based OpenAI Gym \"World Model\" Simulator for Sepsis Treatment","date":"2019-12-15","arxiv_id":"1912.07127","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-dimensional-precision-medicine-from","title":"High dimensional precision medicine from patient-derived xenografts","date":"2019-12-13","arxiv_id":"1912.06667","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-reinforcement-learning-1","title":"Provably Efficient Reinforcement Learning with Aggregated States","date":"2019-12-13","arxiv_id":"1912.06366","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-finite-time-analysis-of-q-learning-with-1","title":"A Finite-Time Analysis of Q-Learning with Neural Network Function Approximation","date":"2019-12-10","arxiv_id":"1912.04511","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-of-information-based-arbitration","title":"Value-of-Information based Arbitration between Model-based and Model-free Control","date":"2019-12-08","arxiv_id":"1912.05453","repositories_listed":0,"syntology":null},{"url":null,"slug":"191202552","title":"Reinforcement Learning with Non-Markovian Rewards","date":"2019-12-05","arxiv_id":"1912.02552","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-q-learning-and-search-with-1","title":"Combining Q-Learning and Search with Amortized Value Estimates","date":"2019-12-05","arxiv_id":"1912.02807","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-switching-system-perspective-and","title":"A Unified Switching System Perspective and O.D.E. Analysis of Q-Learning Algorithms","date":"2019-12-04","arxiv_id":"1912.02270","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-dynamically-coordinate-multi","title":"Learning to Dynamically Coordinate Multi-Robot Teams in Graph Attention Networks","date":"2019-12-04","arxiv_id":"1912.02059","repositories_listed":0,"syntology":null},{"url":null,"slug":"neighborhood-cognition-consistent-multi-agent","title":"Neighborhood Cognition Consistent Multi-Agent Reinforcement Learning","date":"2019-12-03","arxiv_id":"1912.01160","repositories_listed":0,"syntology":null},{"url":null,"slug":"modelling-the-dynamics-of-multiagent-q","title":"Modelling the Dynamics of Multiagent Q-Learning in Repeated Symmetric Games: a Mean Field Theoretic Approach","date":"2019-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-temporal-difference-learning-converges-1","title":"Neural Temporal-Difference Learning Converges to Global Optima","date":"2019-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-q-learning-with-function-1","title":"Provably Efficient Q-learning with Function Approximation via Distribution Shift Error Checking Oracle","date":"2019-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"quadratic-q-network-for-learning-continuous","title":"Quadratic Q-network for Learning Continuous Control for Autonomous Vehicles","date":"2019-11-29","arxiv_id":"1912.00074","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-tutored-reinforcement-learning-an","title":"Control-Tutored Reinforcement Learning: an application to the Herding Problem","date":"2019-11-26","arxiv_id":"1911.11444","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-architecture","title":"A Deep Reinforcement Learning Architecture for Multi-stage Optimal Control","date":"2019-11-25","arxiv_id":"1911.10684","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-modulation-and-coding-based-on","title":"Adaptive Modulation and Coding based on Reinforcement Learning for 5G Networks","date":"2019-11-25","arxiv_id":"1912.04030","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigate-bias-in-face-recognition-using","title":"Mitigate Bias in Face Recognition using Skewness-Aware Reinforcement Learning","date":"2019-11-25","arxiv_id":"1911.10692","repositories_listed":0,"syntology":null},{"url":null,"slug":"which-channel-to-ask-my-question-personalized","title":"Which Channel to Ask My Question? Personalized Customer Service RequestStream Routing using DeepReinforcement Learning","date":"2019-11-24","arxiv_id":"1911.10521","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-drone-mobility-support-using","title":"Efficient Drone Mobility Support Using Reinforcement Learning","date":"2019-11-21","arxiv_id":"1911.09715","repositories_listed":0,"syntology":null}],"record_sha256":"a5d806b1326acb0f43e8d146983f6319e967c1f3e0dbaa9ab37c17df3c75a029","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}