{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/q-learning/papers/18","list_of":"/task/q-learning","task":"Q-Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":18,"pages_in_order":20,"rows_per_page":100,"rows":[1701,1800],"of":1918,"counts":{"archive_papers_tagged":1918,"with_a_code_link":463,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1918,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":102,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":102,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/q-learning","prev":"/task/q-learning/papers/17","next":"/task/q-learning/papers/19","papers":[{"url":null,"slug":"target-based-temporal-difference-learning","title":"Target-Based Temporal Difference Learning","date":"2019-04-24","arxiv_id":"1904.10945","repositories_listed":0,"syntology":null},{"url":null,"slug":"driving-decision-and-control-for-autonomous","title":"Driving Decision and Control for Autonomous Lane Change based on Deep Reinforcement Learning","date":"2019-04-23","arxiv_id":"1904.10171","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-learning-driven-ct-pancreas","title":"Deep Q Learning Driven CT Pancreas Segmentation with Geometry-Aware U-Net","date":"2019-04-19","arxiv_id":"1904.09120","repositories_listed":0,"syntology":null},{"url":null,"slug":"jam-me-if-you-can-defeating-jammer-with-deep","title":"\"Jam Me If You Can'': Defeating Jammer with Deep Dueling Neural Network Architecture and Ambient Backscattering Augmented Communications","date":"2019-04-08","arxiv_id":"1904.03897","repositories_listed":0,"syntology":null},{"url":null,"slug":"patchwork-a-patch-wise-attention-network-for","title":"Patchwork: A Patch-wise Attention Network for Efficient Object Detection and Segmentation in Video Streams","date":"2019-04-03","arxiv_id":"1904.01784","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-cancer-chemotherapy-schedule-a","title":"Personalized Cancer Chemotherapy Schedule: a numerical comparison of performance and robustness in model-based and model-free scheduling methodologies","date":"2019-04-02","arxiv_id":"1904.01200","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-automata-based-q-learning-for","title":"Learning Automata Based Q-learning for Content Placement in Cooperative Caching","date":"2019-03-30","arxiv_id":"1903.06235","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-for-continuous-actions-with-cross","title":"Q-Learning for Continuous Actions with Cross-Entropy Guided Policies","date":"2019-03-25","arxiv_id":"1903.10605","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-characterizing-divergence-in-deep-q","title":"Towards Characterizing Divergence in Deep Q-Learning","date":"2019-03-21","arxiv_id":"1903.08894","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-antenna-tuning-in-heterogeneous","title":"Online Antenna Tuning in Heterogeneous Cellular Networks with Deep Reinforcement Learning","date":"2019-03-15","arxiv_id":"1903.06787","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-multi-agent-reinforcement-learning-with-1","title":"Deep Multi-Agent Reinforcement Learning with Discrete-Continuous Hybrid Action Spaces","date":"2019-03-12","arxiv_id":"1903.04959","repositories_listed":0,"syntology":null},{"url":null,"slug":"successive-over-relaxation-q-learning","title":"Successive Over Relaxation Q-Learning","date":"2019-03-09","arxiv_id":"1903.03812","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-edge-caching-via-reinforcement","title":"Distributed Edge Caching via Reinforcement Learning in Fog Radio Access Networks","date":"2019-02-27","arxiv_id":"1902.10574","repositories_listed":0,"syntology":null},{"url":null,"slug":"unifying-ensemble-methods-for-q-learning-via","title":"Unifying Ensemble Methods for Q-learning via Social Choice Theory","date":"2019-02-27","arxiv_id":"1902.10646","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-and-fast-real-time-resources-slicing","title":"Optimal and Fast Real-time Resources Slicing with Deep Dueling Neural Networks","date":"2019-02-26","arxiv_id":"1902.09696","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributionally-robust-reinforcement","title":"Distributionally Robust Reinforcement Learning","date":"2019-02-23","arxiv_id":"1902.08708","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-the-next-generation-airline-revenue","title":"Autonomous Airline Revenue Management: A Deep Reinforcement Learning Approach to Seat Inventory Control and Overbooking","date":"2019-02-18","arxiv_id":"1902.06824","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-and-short-memory-balancing-in-visual-co","title":"Long and Short Memory Balancing in Visual Co-Tracking using Q-Learning","date":"2019-02-14","arxiv_id":"1902.05211","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-optimal-parametric-q-learning-with","title":"Sample-Optimal Parametric Q-Learning Using Linearly Additive Features","date":"2019-02-13","arxiv_id":"1902.04779","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-best-response-strategies-for-agents","title":"Learning Best Response Strategies for Agents in Ad Exchanges","date":"2019-02-10","arxiv_id":"1902.03588","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-sample-analysis-for-sarsa-and-q","title":"Finite-Sample Analysis for SARSA with Linear Function Approximation","date":"2019-02-06","arxiv_id":"1902.02234","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-theory-of-regularized-markov-decision","title":"A Theory of Regularized Markov Decision Processes","date":"2019-01-31","arxiv_id":"1901.11275","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-with-ucb-exploration-is-sample","title":"Q-learning with UCB Exploration is Sample Efficient for Infinite-Horizon MDP","date":"2019-01-27","arxiv_id":"1901.09311","repositories_listed":0,"syntology":null},{"url":null,"slug":"distillation-strategies-for-proximal-policy","title":"Distillation Strategies for Proximal Policy Optimization","date":"2019-01-23","arxiv_id":"1901.08128","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-markov-decision","title":"Reinforcement Learning of Markov Decision Processes with Peak Constraints","date":"2019-01-23","arxiv_id":"1901.07839","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-goal-directed-reinforcement","title":"Accelerating Goal-Directed Reinforcement Learning by Model Characterization","date":"2019-01-04","arxiv_id":"1901.01977","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-decision-making-in-mixed-agent","title":"Optimal Decision-Making in Mixed-Agent Partially Observable Stochastic Environments via Reinforcement Learning","date":"2019-01-04","arxiv_id":"1901.01325","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-theoretical-analysis-of-deep-q-learning","title":"A Theoretical Analysis of Deep Q-Learning","date":"2019-01-01","arxiv_id":"1901.00137","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-deep-q-learning-for-optimal-execution","title":"Double Deep Q-Learning for Optimal Execution","date":"2018-12-17","arxiv_id":"1812.06600","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-adaptive-caching","title":"Reinforcement Learning for Adaptive Caching with Dynamic Storage Pricing","date":"2018-12-17","arxiv_id":"1812.08593","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-sharing-behaviors-with-arbitrary","title":"Learning Sharing Behaviors with Arbitrary Numbers of Agents","date":"2018-12-10","arxiv_id":"1812.04145","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-multilayer-optical-film-optimal-method","title":"A new multilayer optical film optimal method based on deep q-learning","date":"2018-12-07","arxiv_id":"1812.02873","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-deep-q-learning-with-demonstration","title":"Active Deep Q-learning with Demonstration","date":"2018-12-06","arxiv_id":"1812.02632","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-delusional-q-learning-and-value-iteration","title":"Non-delusional Q-learning and value-iteration","date":"2018-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"emergence-of-addictive-behaviors-in","title":"Emergence of Addictive Behaviors in Reinforcement Learning Agents","date":"2018-11-14","arxiv_id":"1811.05590","repositories_listed":0,"syntology":null},{"url":null,"slug":"managing-app-install-ad-campaigns-in-rtb-a-q","title":"Managing App Install Ad Campaigns in RTB: A Q-Learning Approach","date":"2018-11-11","arxiv_id":"1811.04475","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-green","title":"Deep Reinforcement Learning for Green Security Games with Real-Time Information","date":"2018-11-06","arxiv_id":"1811.02483","repositories_listed":0,"syntology":null},{"url":null,"slug":"quasi-newton-optimization-in-deep-q-learning","title":"Deep Reinforcement Learning via L-BFGS Optimization","date":"2018-11-06","arxiv_id":"1811.02693","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-dynamic-model","title":"Reinforcement Learning based Dynamic Model Selection for Short-Term Load Forecasting","date":"2018-11-05","arxiv_id":"1811.01846","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-dynamic-oracle-for-dependency","title":"Approximate Dynamic Oracle for Dependency Parsing with Reinforcement Learning","date":"2018-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"structure-learning-of-deep-neural-networks","title":"Structure Learning of Deep Neural Networks with Q-Learning","date":"2018-10-31","arxiv_id":"1810.13155","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributive-dynamic-spectrum-access-through","title":"Distributive Dynamic Spectrum Access through Deep Reinforcement Learning: A Reservoir Computing Based Approach","date":"2018-10-28","arxiv_id":"1810.11758","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-negotiating-behavior-between-cars-in","title":"Learning Negotiating Behavior Between Cars in Intersections using Deep Q-Learning","date":"2018-10-24","arxiv_id":"1810.10469","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-based-1","title":"Multi-Agent Reinforcement Learning Based Resource Allocation for UAV Networks","date":"2018-10-24","arxiv_id":"1810.10408","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-the-best-design-parameters-for","title":"Finding the best design parameters for optical nanostructures using reinforcement learning","date":"2018-10-18","arxiv_id":"1810.10964","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-sketch-with-deep-q-networks-and","title":"Learning to Sketch with Deep Q Networks and Demonstrated Strokes","date":"2018-10-14","arxiv_id":"1810.05977","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-reason","title":"Learning to Reason","date":"2018-10-12","arxiv_id":"1810.05315","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-evolutionary-learning-method","title":"Reinforcement Evolutionary Learning Method for self-learning","date":"2018-10-07","arxiv_id":"1810.03198","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-r","title":"Reinforcement Learning in R","date":"2018-09-29","arxiv_id":"1810.00240","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-convergent-variant-of-the-boltzmann-softmax","title":"A Convergent Variant of the Boltzmann Softmax Operator in Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerated-value-iteration-via-anderson","title":"Accelerated Value Iteration via Anderson Mixing","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"convergent-reinforcement-learning-with","title":"Convergent Reinforcement Learning with Function Approximation: A Bilevel Optimization Perspective","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-policies-using-inverse-rewards-for","title":"Hybrid Policies Using Inverse Rewards for Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-wisdom-of-the-crowd-reliable-deep","title":"The wisdom of the crowd: reliable deep reinforcement learning through ensembles of Q-functions","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"what-would-pi-do-imitation-learning-via-off","title":"What Would pi* Do?: Imitation Learning via Off-Policy Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-through-probing-a-decentralized","title":"Learning through Probing: a decentralized reinforcement learning architecture for social dilemmas","date":"2018-09-26","arxiv_id":"1809.10007","repositories_listed":0,"syntology":null},{"url":null,"slug":"floyd-warshall-reinforcement-learning","title":"Floyd-Warshall Reinforcement Learning: Learning from Past Experiences to Reach New Goals","date":"2018-09-25","arxiv_id":"1809.09318","repositories_listed":0,"syntology":null},{"url":null,"slug":"target-transfer-q-learning-and-its","title":"Target Transfer Q-Learning and Its Convergence Analysis","date":"2018-09-21","arxiv_id":"1809.08923","repositories_listed":0,"syntology":null},{"url":null,"slug":"hidden-markov-model-estimation-based-q","title":"Hidden Markov Model Estimation-Based Q-learning for Partially Observable Markov Decision Process","date":"2018-09-17","arxiv_id":"1809.06401","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-matrix-momentum-stochastic","title":"Optimal Matrix Momentum Stochastic Approximation and Applications to Q-learning","date":"2018-09-17","arxiv_id":"1809.06277","repositories_listed":0,"syntology":null},{"url":null,"slug":"directed-exploration-in-pac-model-free","title":"Directed Exploration in PAC Model-Free Reinforcement Learning","date":"2018-08-31","arxiv_id":"1808.10552","repositories_listed":0,"syntology":null},{"url":null,"slug":"marl-fwc-optimal-coordination-of-freeway","title":"MARL-FWC: Optimal Coordination of Freeway Traffic Control Measures","date":"2018-08-27","arxiv_id":"1808.09806","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-derivation-of-formulas-using","title":"Automatic Derivation Of Formulas Using Reforcement Learning","date":"2018-08-15","arxiv_id":"1808.04946","repositories_listed":0,"syntology":null},{"url":null,"slug":"robbins-monro-conditions-for-persistent","title":"Robbins-Monro conditions for persistent exploration learning strategies","date":"2018-08-01","arxiv_id":"1808.00245","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-to-target","title":"A Reinforcement Learning Approach to Target Tracking in a Camera Network","date":"2018-07-26","arxiv_id":"1807.10336","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-bayesian-reinforcement-learning","title":"Variational Bayesian Reinforcement Learning with Regret Bounds","date":"2018-07-25","arxiv_id":"1807.09647","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerated-structure-aware-reinforcement","title":"Accelerated Structure-Aware Reinforcement Learning for Delay-Sensitive Energy Harvesting Wireless Sensors","date":"2018-07-22","arxiv_id":"1807.08315","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-linear-complexity-reinforcement","title":"Discrete linear-complexity reinforcement learning in continuous action spaces for Q-learning algorithms","date":"2018-07-16","arxiv_id":"1807.06957","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-summarisation-by-classification-with","title":"Video Summarisation by Classification with Deep Reinforcement Learning","date":"2018-07-09","arxiv_id":"1807.03089","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-against-nature-causal-discovery-for","title":"Playing against Nature: causal discovery for decision making under uncertainty","date":"2018-07-03","arxiv_id":"1807.01268","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-coordinate-with-coordination","title":"Learning to Coordinate with Coordination Graphs in Repeated Single-Stage Multi-Agent Decision Problems","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-explore-via-meta-policy-gradient","title":"Learning to Explore via Meta-Policy Gradient","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"many-goals-reinforcement-learning","title":"Many-Goals Reinforcement Learning","date":"2018-06-22","arxiv_id":"1806.09605","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-using-augmented-neural","title":"Reinforcement Learning using Augmented Neural Networks","date":"2018-06-20","arxiv_id":"1806.07692","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-learning-for-3d-point-cloud-based","title":"Action Learning for 3D Point Cloud Based Organ Segmentation","date":"2018-06-14","arxiv_id":"1806.05724","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-formation-of-the-structure-of","title":"Automatic formation of the structure of abstract machines in hierarchical reinforcement learning with state clustering","date":"2018-06-13","arxiv_id":"1806.05292","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-advantage-actor-critic","title":"Distributional Advantage Actor-Critic","date":"2018-06-10","arxiv_id":"1806.06914","repositories_listed":0,"syntology":null},{"url":null,"slug":"fidelity-based-probabilistic-q-learning-for","title":"Fidelity-based Probabilistic Q-learning for Control of Quantum Systems","date":"2018-06-08","arxiv_id":"1806.03145","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-finite-time-analysis-of-temporal-difference","title":"A Finite Time Analysis of Temporal Difference Learning With Linear Function Approximation","date":"2018-06-06","arxiv_id":"1806.02450","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperparameter-optimization-for-tracking-with","title":"Hyperparameter Optimization for Tracking With Continuous Deep Q-Learning","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"depth-and-nonlinearity-induce-implicit","title":"Depth and nonlinearity induce implicit exploration for RL","date":"2018-05-29","arxiv_id":"1805.11711","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-clustering-with-deep-q-learning","title":"Hierarchical clustering with deep Q-learning","date":"2018-05-28","arxiv_id":"1805.10900","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-self-imitating-diverse-policies","title":"Learning Self-Imitating Diverse Policies","date":"2018-05-25","arxiv_id":"1805.10309","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-simple-exploration-is-sample-efficient","title":"When Simple Exploration is Sample Efficient: Identifying Sufficient Conditions for Random Exploration to Yield PAC RL Algorithms","date":"2018-05-23","arxiv_id":"1805.09045","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-sampling-policies-for-domain","title":"Learning Sampling Policies for Domain Adaptation","date":"2018-05-19","arxiv_id":"1805.07641","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithmic-trading-with-fitted-q-iteration","title":"Algorithmic Trading with Fitted Q Iteration and Heston Model","date":"2018-05-18","arxiv_id":"1805.07478","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-approximation-for-risk-aware","title":"Stochastic Approximation for Risk-aware Markov Decision Processes","date":"2018-05-11","arxiv_id":"1805.04238","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-and-learning-with-stochastic-action","title":"Planning and Learning with Stochastic Action Sets","date":"2018-05-07","arxiv_id":"1805.02363","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hybrid-q-learning-sine-cosine-based","title":"A Hybrid Q-Learning Sine-Cosine-based Strategy for Addressing the Combinatorial Test Suite Minimization Problem","date":"2018-04-27","arxiv_id":"1805.00873","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiagent-soft-q-learning","title":"Multiagent Soft Q-Learning","date":"2018-04-25","arxiv_id":"1804.09817","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-projective-simulation-in","title":"Benchmarking projective simulation in navigation problems","date":"2018-04-23","arxiv_id":"1804.08607","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-distribution-aware-sampling-for-deep-q","title":"State Distribution-aware Sampling for Deep Q-learning","date":"2018-04-23","arxiv_id":"1804.08619","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-co-training","title":"Reinforced Co-Training","date":"2018-04-17","arxiv_id":"1804.06035","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-augmentation-transformations-for-risk","title":"State-Augmentation Transformations for Risk-Sensitive Reinforcement Learning","date":"2018-04-16","arxiv_id":"1804.05950","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-modular-reinforcement-learning","title":"Hierarchical Modular Reinforcement Learning Method and Knowledge Acquisition of State-Action Rule for Multi-target Problem","date":"2018-04-08","arxiv_id":"1804.02698","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-maximizing-exploration-with-a","title":"Information Maximizing Exploration with a Latent Dynamics Model","date":"2018-04-04","arxiv_id":"1804.01238","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-learning-of-interactive-spoken-content","title":"Joint Learning of Interactive Spoken Content Retrieval and Trainable User Simulator","date":"2018-04-01","arxiv_id":"1804.00318","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-gradient-deep-q-learning","title":"Natural Gradient Deep Q-learning","date":"2018-03-20","arxiv_id":"1803.07482","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-explore-with-meta-policy-gradient","title":"Learning to Explore with Meta-Policy Gradient","date":"2018-03-13","arxiv_id":"1803.05044","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-armed-bandits-for-correlated-markovian","title":"Multi-Armed Bandits for Correlated Markovian Environments with Smoothed Reward Feedback","date":"2018-03-11","arxiv_id":"1803.04008","repositories_listed":0,"syntology":null}],"record_sha256":"955919fcf5cf59ca00c8f1f74f1d3d6fb5369f74b9fb869974870080b5596e4e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}