{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/q-learning/papers/19","list_of":"/task/q-learning","task":"Q-Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":19,"pages_in_order":20,"rows_per_page":100,"rows":[1801,1900],"of":1918,"counts":{"archive_papers_tagged":1918,"with_a_code_link":463,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1918,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":102,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":102,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/q-learning","prev":"/task/q-learning/papers/18","next":"/task/q-learning/papers/20","papers":[{"url":null,"slug":"sa-iga-a-multiagent-reinforcement-learning","title":"SA-IGA: A Multiagent Reinforcement Learning Method Towards Socially Optimal Outcomes","date":"2018-03-08","arxiv_id":"1803.03021","repositories_listed":0,"syntology":null},{"url":null,"slug":"smoothed-action-value-functions-for-learning","title":"Smoothed Action Value Functions for Learning Gaussian Policies","date":"2018-03-06","arxiv_id":"1803.02348","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-cp-learning-action-values-for-cooperative","title":"Q-CP: Learning Action Values for Cooperative Planning","date":"2018-03-01","arxiv_id":"1803.00297","repositories_listed":0,"syntology":null},{"url":null,"slug":"variance-reduction-methods-for-sublinear","title":"Variance Reduction Methods for Sublinear Reinforcement Learning","date":"2018-02-26","arxiv_id":"1802.09184","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-difference-models-model-free-deep-rl","title":"Temporal Difference Models: Model-Free Deep RL for Model-Based Control","date":"2018-02-25","arxiv_id":"1802.09081","repositories_listed":0,"syntology":null},{"url":null,"slug":"weighted-double-deep-multiagent-reinforcement","title":"Weighted Double Deep Multiagent Reinforcement Learning in Stochastic Cooperative Environments","date":"2018-02-23","arxiv_id":"1802.08534","repositories_listed":0,"syntology":null},{"url":null,"slug":"prioritized-sweeping-neural-dynaq-with","title":"Prioritized Sweeping Neural DynaQ with Multiple Predecessors, and Hippocampal Replays","date":"2018-02-15","arxiv_id":"1802.05594","repositories_listed":0,"syntology":null},{"url":"/paper/m-walk-learning-to-walk-over-graphs-using","slug":"m-walk-learning-to-walk-over-graphs-using","title":"M-Walk: Learning to Walk over Graphs using Monte Carlo Tree Search","date":"2018-02-12","arxiv_id":"1802.04394","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-with-nearest-neighbors","title":"Q-learning with Nearest Neighbors","date":"2018-02-12","arxiv_id":"1802.03900","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-two-player-stochastic-games-with","title":"Balancing Two-Player Stochastic Games with Soft Q-Learning","date":"2018-02-09","arxiv_id":"1802.03216","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-using-capsules-in","title":"Deep Reinforcement Learning using Capsules in Advanced Game Environments","date":"2018-01-29","arxiv_id":"1801.09597","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-qlbs-q-learner-goes-nuqlear-fitted-q","title":"The QLBS Q-Learner Goes NuQLear: Fitted Q Iteration, Inverse RL, and Option Portfolios","date":"2018-01-17","arxiv_id":"1801.06077","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-fuzzing","title":"Deep Reinforcement Fuzzing","date":"2018-01-14","arxiv_id":"1801.04589","repositories_listed":0,"syntology":null},{"url":null,"slug":"trading-the-twitter-sentiment-with","title":"Trading the Twitter Sentiment with Reinforcement Learning","date":"2018-01-07","arxiv_id":"1801.02243","repositories_listed":0,"syntology":null},{"url":null,"slug":"faster-deep-q-learning-using-neural-episodic","title":"Faster Deep Q-learning using Neural Episodic Control","date":"2018-01-06","arxiv_id":"1801.01968","repositories_listed":0,"syntology":null},{"url":null,"slug":"screenernet-learning-self-paced-curriculum","title":"ScreenerNet: Learning Self-Paced Curriculum for Deep Neural Networks","date":"2018-01-03","arxiv_id":"1801.00904","repositories_listed":0,"syntology":null},{"url":null,"slug":"vizdoom-drqn-with-prioritized-experience","title":"ViZDoom: DRQN with Prioritized Experience Replay, Double-Q Learning, & Snapshot Ensembling","date":"2018-01-03","arxiv_id":"1801.01000","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-vehicle-fleet-coordination-with","title":"Autonomous Vehicle Fleet Coordination With Deep Reinforcement Learning","date":"2018-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"avoiding-catastrophic-states-with-intrinsic","title":"Avoiding Catastrophic States with Intrinsic Fear","date":"2018-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-gaussian-policies-from-smoothed","title":"Learning Gaussian Policies from Smoothed Action Value Functions","date":"2018-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"representing-entropy-a-short-proof-of-the","title":"Representing Entropy : A short proof of the equivalence between soft Q-learning and policy gradients","date":"2018-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"td-learning-with-constrained-gradients","title":"TD Learning with Constrained Gradients","date":"2018-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sbeed-convergent-reinforcement-learning-with","title":"SBEED: Convergent Reinforcement Learning with Nonlinear Function Approximation","date":"2017-12-29","arxiv_id":"1712.10285","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-short-variational-proof-of-equivalence","title":"A short variational proof of equivalence between policy gradients and soft Q learning","date":"2017-12-22","arxiv_id":"1712.08650","repositories_listed":0,"syntology":null},{"url":null,"slug":"scale-invariant-temporal-history-sith-optimal","title":"Scale-invariant temporal history (SITH): optimal slicing of the past in an uncertain world","date":"2017-12-19","arxiv_id":"1712.07165","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-deep-reinforcement-learning","title":"Towards a Deep Reinforcement Learning Approach for Tower Line Wars","date":"2017-12-17","arxiv_id":"1712.06180","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-deep-reinforcement-learning-with","title":"Robust Deep Reinforcement Learning with Adversarial Attacks","date":"2017-12-11","arxiv_id":"1712.03632","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-primal-dual-reinforcement-learning","title":"Deep Primal-Dual Reinforcement Learning: Accelerating Actor-Critic using Bellman Duality","date":"2017-12-07","arxiv_id":"1712.02467","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-lda-uncovering-latent-patterns-in-text","title":"Q-LDA: Uncovering Latent Patterns in Text-based Sequential Decision Processes","date":"2017-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"zap-q-learning","title":"Zap Q-Learning","date":"2017-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-q-learning-for-visual-vocabulary","title":"Curriculum Q-Learning for Visual Vocabulary Acquisition","date":"2017-11-29","arxiv_id":"1711.10837","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-algorithm-for","title":"A reinforcement learning algorithm for building collaboration in multi-agent systems","date":"2017-11-28","arxiv_id":"1711.10574","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-network-based-reinforcement-learning","title":"Neural Network Based Reinforcement Learning for Audio-Visual Gaze Control in Human-Robot Interaction","date":"2017-11-18","arxiv_id":"1711.06834","repositories_listed":0,"syntology":null},{"url":null,"slug":"bbq-networks-efficient-exploration-in-deep","title":"BBQ-Networks: Efficient Exploration in Deep Reinforcement Learning for Task-Oriented Dialogue Systems","date":"2017-11-15","arxiv_id":"1711.05715","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-decision-making-framework-for","title":"A unified decision making framework for supply and demand management in microgrid networks","date":"2017-11-14","arxiv_id":"1711.05078","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-q-and-q-unifying-reinforcement","title":"Double Q($σ$) and Q($σ, λ$): Unifying Reinforcement Learning Control Algorithms","date":"2017-11-05","arxiv_id":"1711.01569","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-framework","title":"Deep Reinforcement Learning: Framework, Applications, and Embedded Implementations","date":"2017-10-10","arxiv_id":"1710.03792","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-q-walk-for-learning-vector","title":"Supervised Q-walk for Learning Vector Representation of Nodes in Networks","date":"2017-10-03","arxiv_id":"1710.00978","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-reinforcement-learning-mechanism-for","title":"A Simple Reinforcement Learning Mechanism for Resource Allocation in LTE-A Networks with Markov Decision Process and Q-Learning","date":"2017-09-27","arxiv_id":"1709.09312","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-optimal-online-method-of-selecting-source","title":"An Optimal Online Method of Selecting Source Policies for Reinforcement Learning","date":"2017-09-24","arxiv_id":"1709.08201","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-search-through-a3c-reinforcement","title":"Improving Search through A3C Reinforcement Learning based Conversational Agent","date":"2017-09-17","arxiv_id":"1709.05638","repositories_listed":0,"syntology":null},{"url":null,"slug":"bibi-system-description-building-with-cnns","title":"BIBI System Description: Building with CNNs and Breaking with Deep Reinforcement Learning","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"constructing-narrative-using-a-generative","title":"Constructing narrative using a generative model and continuous action policies","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-q-learning-for-minimizing-demand","title":"Multi-Agent Q-Learning for Minimizing Demand-Supply Power Deficit in Microgrids","date":"2017-08-25","arxiv_id":"1708.07732","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-reinforcement-learning-agents","title":"Investigating Reinforcement Learning Agents for Continuous State Space Environments","date":"2017-08-08","arxiv_id":"1708.02378","repositories_listed":0,"syntology":null},{"url":null,"slug":"guiding-reinforcement-learning-exploration","title":"Guiding Reinforcement Learning Exploration Using Natural Language","date":"2017-07-26","arxiv_id":"1707.08616","repositories_listed":0,"syntology":null},{"url":null,"slug":"empirical-evaluation-of-a-q-learning","title":"Empirical evaluation of a Q-Learning Algorithm for Model-free Autonomous Soaring","date":"2017-07-18","arxiv_id":"1707.05668","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-line-building-energy-optimization-using","title":"On-line Building Energy Optimization using Deep Reinforcement Learning","date":"2017-07-18","arxiv_id":"1707.05878","repositories_listed":0,"syntology":null},{"url":null,"slug":"fastest-convergence-for-q-learning","title":"Fastest Convergence for Q-learning","date":"2017-07-12","arxiv_id":"1707.03770","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-learning-for-self-organizing-networks","title":"Deep Q-Learning for Self-Organizing Networks Fault Management and Radio Performance Improvement","date":"2017-07-10","arxiv_id":"1707.02329","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-algorithm-for-volte-closed-loop","title":"Q-Learning Algorithm for VoLTE Closed-Loop Power Control in Indoor Small Cells","date":"2017-07-10","arxiv_id":"1707.03269","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-under-model-mismatch","title":"Reinforcement Learning under Model Mismatch","date":"2017-06-15","arxiv_id":"1706.04711","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-learn-from-noisy-web-videos","title":"Learning to Learn from Noisy Web Videos","date":"2017-06-09","arxiv_id":"1706.02884","repositories_listed":0,"syntology":null},{"url":null,"slug":"ucb-exploration-via-q-ensembles","title":"UCB Exploration via Q-Ensembles","date":"2017-06-05","arxiv_id":"1706.01502","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-factor-policies-and-action-value","title":"Learning to Factor Policies and Action-Value Functions: Factored Action Space Representations for Deep Reinforcement learning","date":"2017-05-20","arxiv_id":"1705.07269","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-reinforcement-learning","title":"A Comparison of Reinforcement Learning Techniques for Fuzzy Cloud Auto-Scaling","date":"2017-05-19","arxiv_id":"1705.07114","repositories_listed":0,"syntology":null},{"url":null,"slug":"identification-and-off-policy-learning-of","title":"Identification and Off-Policy Learning of Multiple Objectives Using Adaptive Clustering","date":"2017-05-17","arxiv_id":"1705.06342","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-represent-haptic-feedback-for","title":"Learning to Represent Haptic Feedback for Partially-Observable Tasks","date":"2017-05-17","arxiv_id":"1705.06243","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-hard-alignments-with-variational","title":"Learning Hard Alignments with Variational Inference","date":"2017-05-16","arxiv_id":"1705.05524","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-sequential-prediction-of-continuous","title":"Discrete Sequential Prediction of Continuous Actions for Deep RL","date":"2017-05-14","arxiv_id":"1705.05035","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-episodic-value-iteration-for-model-based","title":"Deep Episodic Value Iteration for Model-based Meta-Reinforcement Learning","date":"2017-05-09","arxiv_id":"1705.03562","repositories_listed":0,"syntology":null},{"url":null,"slug":"equivalence-between-policy-gradients-and-soft","title":"Equivalence Between Policy Gradients and Soft Q-Learning","date":"2017-04-21","arxiv_id":"1704.06440","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-external","title":"Reinforcement Learning with External Knowledge and Two-Stage Q-functions for Predicting Popular Reddit Threads","date":"2017-04-20","arxiv_id":"1704.06217","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-deep-reinforcement-learning","title":"Data-efficient Deep Reinforcement Learning for Dexterous Manipulation","date":"2017-04-10","arxiv_id":"1704.03073","repositories_listed":0,"syntology":null},{"url":null,"slug":"pseudorehearsal-in-value-function","title":"Pseudorehearsal in value function approximation","date":"2017-03-21","arxiv_id":"1703.07075","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-learning-for-offloading-and","title":"Online Learning for Offloading and Autoscaling in Energy Harvesting Mobile Edge Computing","date":"2017-03-17","arxiv_id":"1703.06060","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-step-reinforcement-learning-a-unifying","title":"Multi-step Reinforcement Learning: A Unifying Algorithm","date":"2017-03-03","arxiv_id":"1703.01327","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-control-for-air-hockey-striking","title":"Learning Control for Air Hockey Striking using Deep Reinforcement Learning","date":"2017-02-26","arxiv_id":"1702.08074","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-deep-reinforcement-learning-for-1","title":"Collaborative Deep Reinforcement Learning for Joint Object Search","date":"2017-02-18","arxiv_id":"1702.05573","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-game-imitation-deep-supervised","title":"The Game Imitation: Deep Supervised Convolutional Networks for Quick Video Game AI","date":"2017-02-18","arxiv_id":"1702.05663","repositories_listed":0,"syntology":null},{"url":null,"slug":"fpga-architecture-for-deep-learning-and-its","title":"FPGA Architecture for Deep Learning and its application to Planetary Robotics","date":"2017-01-26","arxiv_id":"1701.07543","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-predict-where-to-look-in","title":"Learning to predict where to look in interactive environments using deep recurrent q-learning","date":"2016-12-17","arxiv_id":"1612.05753","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-differentiable-physics-engine-for-deep","title":"A Differentiable Physics Engine for Deep Learning in Robotics","date":"2016-11-05","arxiv_id":"1611.01652","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-policy-gradient-and-q-learning","title":"Combining policy gradient and Q-learning","date":"2016-11-05","arxiv_id":"1611.01626","repositories_listed":0,"syntology":null},{"url":null,"slug":"combating-reinforcement-learnings-sisyphean","title":"Combating Reinforcement Learning's Sisyphean Curse with Intrinsic Fear","date":"2016-11-03","arxiv_id":"1611.01211","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-a-deep-reinforcement-learning-agent-for","title":"Using a Deep Reinforcement Learning Agent for Traffic Signal Control","date":"2016-11-03","arxiv_id":"1611.01142","repositories_listed":0,"syntology":null},{"url":null,"slug":"internet-of-things-applications-animal","title":"Internet of Things Applications: Animal Monitoring with Unmanned Aerial Vehicle","date":"2016-10-17","arxiv_id":"1610.05287","repositories_listed":0,"syntology":null},{"url":null,"slug":"modelling-stock-market-investors-as","title":"Modelling Stock-market Investors as Reinforcement Learning Agents [Correction]","date":"2016-09-20","arxiv_id":"1609.06086","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-spoken-content-retrieval-by-deep","title":"Interactive Spoken Content Retrieval by Deep Reinforcement Learning","date":"2016-09-16","arxiv_id":"1609.05234","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-simulation-for-robot-arm-control-with-deep","title":"3D Simulation for Robot Arm Control with Deep Q-Learning","date":"2016-09-13","arxiv_id":"1609.03759","repositories_listed":0,"syntology":null},{"url":null,"slug":"episodic-exploration-for-deep-deterministic","title":"Episodic Exploration for Deep Deterministic Policies: An Application to StarCraft Micromanagement Tasks","date":"2016-09-10","arxiv_id":"1609.02993","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-exit-configuration-of-mesoscopic","title":"Multi Exit Configuration of Mesoscopic Pedestrian Simulation","date":"2016-09-06","arxiv_id":"1609.01475","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-with-basic-emotions","title":"Q-Learning with Basic Emotions","date":"2016-09-06","arxiv_id":"1609.01468","repositories_listed":0,"syntology":null},{"url":null,"slug":"bbq-networks-efficient-exploration-in-deep-1","title":"BBQ-Networks: Efficient Exploration in Deep Reinforcement Learning for Task-Oriented Dialogue Systems","date":"2016-08-17","arxiv_id":"1608.05081","repositories_listed":0,"syntology":null},{"url":null,"slug":"neurohex-a-deep-q-learning-hex-agent","title":"Neurohex: A Deep Q-learning Hex Agent","date":"2016-04-24","arxiv_id":"1604.07097","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approach-for-real-time","title":"Reinforcement Learning approach for Real Time Strategy Games Battle city and S3","date":"2016-02-16","arxiv_id":"1602.04936","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-deep-q-learning-to-control-optimization","title":"Using Deep Q-Learning to Control Optimization Hyperparameters","date":"2016-02-12","arxiv_id":"1602.04062","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-networks-for-binary-vector-actions","title":"Q-Networks for Binary Vector Actions","date":"2015-12-04","arxiv_id":"1512.01332","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-attention","title":"Deep Reinforcement Learning with Attention for Slate Markov Decision Processes with High-Dimensional States and Actions","date":"2015-12-03","arxiv_id":"1512.01124","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-search-rescue-via-online-multi-task","title":"Robotic Search & Rescue via Online Multi-task Reinforcement Learning","date":"2015-11-29","arxiv_id":"1511.08967","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-phase-q-learning-for-bidding-based","title":"Two Phase $Q-$learning for Bidding-based Vehicle Sharing","date":"2015-09-29","arxiv_id":"1509.08932","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimization-of-anemia-treatment-in","title":"Optimization of anemia treatment in hemodialysis patients via reinforcement learning","date":"2015-09-14","arxiv_id":"1509.03977","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-deep-q-learning","title":"Distributed Deep Q-Learning","date":"2015-08-18","arxiv_id":"1508.04186","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-prediction-markets-for-online","title":"Artificial Prediction Markets for Online Prediction of Continuous Variables-A Preliminary Report","date":"2015-08-11","arxiv_id":"1508.02681","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-transfer-learning-in-reinforcement","title":"Online Transfer Learning in Reinforcement Learning Domains","date":"2015-07-02","arxiv_id":"1507.00436","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-q-learning-for-stochastic-teams","title":"Decentralized Q-Learning for Stochastic Teams and Games","date":"2015-06-25","arxiv_id":"1506.07924","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-crm-control-via-clv-approximation","title":"Autonomous CRM Control via CLV Approximation with Deep Reinforcement Learning in Discrete and Continuous Action Space","date":"2015-04-08","arxiv_id":"1504.01840","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-sharing-for-multiple-sensor-nodes-with","title":"Energy Sharing for Multiple Sensor Nodes with Finite Buffers","date":"2015-03-17","arxiv_id":"1503.04964","repositories_listed":0,"syntology":null},{"url":null,"slug":"correct-by-synthesis-reinforcement-learning","title":"Correct-by-synthesis reinforcement learning with temporal logic constraints","date":"2015-03-05","arxiv_id":"1503.01793","repositories_listed":0,"syntology":null},{"url":null,"slug":"empirical-q-value-iteration","title":"Empirical Q-Value Iteration","date":"2014-11-30","arxiv_id":"1412.0180","repositories_listed":0,"syntology":null}],"record_sha256":"078caa52e81aaf975ef6a26d0e50070f2ecb1c8667da7ae98baaf13145a097ae","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}