{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/q-learning/papers/5","list_of":"/task/q-learning","task":"Q-Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":20,"rows_per_page":100,"rows":[401,500],"of":1918,"counts":{"archive_papers_tagged":1918,"with_a_code_link":463,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1918,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":102,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":102,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/q-learning","prev":"/task/q-learning/papers/4","next":"/task/q-learning/papers/6","papers":[{"url":"/paper/reinforcement-learning-with-low-complexity","slug":"reinforcement-learning-with-low-complexity","title":"Reinforcement Learning with Low-Complexity Liquid State Machines","date":"2019-06-04","arxiv_id":"1906.01695","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-with-low-complexity#ran","syntology_url":"https://syntology.ai/paper/1906.01695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.01695"}},"official":{"repos":["wponghiran/lsm-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/finite-time-analysis-of-q-learning-with","slug":"finite-time-analysis-of-q-learning-with","title":"Finite-Sample Analysis of Nonlinear Stochastic Approximation with Applications in Reinforcement Learning","date":"2019-05-27","arxiv_id":"1905.11425","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finite-time-analysis-of-q-learning-with#ran","syntology_url":"https://syntology.ai/paper/1905.11425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.11425"}},"official":null}},{"url":"/paper/a-kernel-loss-for-solving-the-bellman","slug":"a-kernel-loss-for-solving-the-bellman","title":"A Kernel Loss for Solving the Bellman Equation","date":"2019-05-25","arxiv_id":"1905.10506","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-symmetric-reward-noising-for","slug":"adaptive-symmetric-reward-noising-for","title":"Adaptive Symmetric Reward Noising for Reinforcement Learning","date":"2019-05-24","arxiv_id":"1905.10144","repositories_listed":1,"syntology":null},{"url":"/paper/neural-temporal-difference-learning-converges","slug":"neural-temporal-difference-learning-converges","title":"Neural Temporal-Difference and Q-Learning Provably Converge to Global Optima","date":"2019-05-24","arxiv_id":"1905.10027","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-parameter","slug":"deep-reinforcement-learning-based-parameter","title":"Deep Reinforcement Learning Based Parameter Control in Differential Evolution","date":"2019-05-20","arxiv_id":"1905.08006","repositories_listed":1,"syntology":null},{"url":"/paper/qbso-fs-a-reinforcement-learning-based-bee","slug":"qbso-fs-a-reinforcement-learning-based-bee","title":"QBSO-FS: A Reinforcement Learning Based Bee Swarm Optimization Metaheuristic for Feature Selection","date":"2019-05-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/stochastic-approximation-with-cone","slug":"stochastic-approximation-with-cone","title":"Stochastic approximation with cone-contractive operators: Sharp $\\ell_\\infty$-bounds for $Q$-learning","date":"2019-05-15","arxiv_id":"1905.06265","repositories_listed":1,"syntology":null},{"url":"/paper/deep-ordinal-reinforcement-learning","slug":"deep-ordinal-reinforcement-learning","title":"Deep Ordinal Reinforcement Learning","date":"2019-05-06","arxiv_id":"1905.02005","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-model-free-reinforcement-learning-1","slug":"efficient-model-free-reinforcement-learning-1","title":"Efficient Model-free Reinforcement Learning in Metric Spaces","date":"2019-05-01","arxiv_id":"1905.00475","repositories_listed":1,"syntology":null},{"url":"/paper/deep-q-learning-for-nash-equilibria-nash-dqn","slug":"deep-q-learning-for-nash-equilibria-nash-dqn","title":"Deep Q-Learning for Nash Equilibria: Nash-DQN","date":"2019-04-23","arxiv_id":"1904.10554","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-dynamic-boltzmann","slug":"reinforcement-learning-with-dynamic-boltzmann","title":"Reinforcement Learning with Dynamic Boltzmann Softmax Updates","date":"2019-03-14","arxiv_id":"1903.05926","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-deep-reinforcement-learning-for-2","slug":"multi-agent-deep-reinforcement-learning-for-2","title":"Multi-Agent Deep Reinforcement Learning for Large-scale Traffic Signal Control","date":"2019-03-11","arxiv_id":"1903.04527","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-deep-reinforcement-learning-for-2#ran","syntology_url":"https://syntology.ai/paper/1903.04527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04527"}},"official":{"repos":["cts198859/deeprl_signal_control"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/diagnosing-bottlenecks-in-deep-q-learning","slug":"diagnosing-bottlenecks-in-deep-q-learning","title":"Diagnosing Bottlenecks in Deep Q-learning Algorithms","date":"2019-02-26","arxiv_id":"1902.10250","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diagnosing-bottlenecks-in-deep-q-learning#ran","syntology_url":"https://syntology.ai/paper/1902.10250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.10250"}},"official":null}},{"url":"/paper/heuristics-answer-set-programming-and-markov","slug":"heuristics-answer-set-programming-and-markov","title":"Heuristics, Answer Set Programming and Markov Decision Process for Solving a Set of Spatial Puzzles","date":"2019-02-16","arxiv_id":"1903.03411","repositories_listed":1,"syntology":null},{"url":"/paper/augmenting-learning-components-for-safety-in","slug":"augmenting-learning-components-for-safety-in","title":"Dynamic-Weighted Simplex Strategy for Learning Enabled Cyber Physical Systems","date":"2019-02-06","arxiv_id":"1902.02432","repositories_listed":1,"syntology":null},{"url":"/paper/private-q-learning-with-functional-noise-in","slug":"private-q-learning-with-functional-noise-in","title":"Privacy-preserving Q-Learning with Functional Noise in Continuous State Spaces","date":"2019-01-30","arxiv_id":"1901.10634","repositories_listed":1,"syntology":null},{"url":"/paper/making-deep-q-learning-methods-robust-to-time","slug":"making-deep-q-learning-methods-robust-to-time","title":"Making Deep Q-learning methods robust to time discretization","date":"2019-01-28","arxiv_id":"1901.09732","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-deep-q-learning-methods-robust-to-time#ran","syntology_url":"https://syntology.ai/paper/1901.09732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09732"}},"official":null}},{"url":"/paper/provably-efficient-rl-with-rich-observations","slug":"provably-efficient-rl-with-rich-observations","title":"Provably efficient RL with Rich Observations via Latent State Decoding","date":"2019-01-25","arxiv_id":"1901.09018","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/provably-efficient-rl-with-rich-observations#ran","syntology_url":"https://syntology.ai/paper/1901.09018","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09018"}},"official":{"repos":["Microsoft/StateDecoding"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/combinational-q-learning-for-dou-di-zhu","slug":"combinational-q-learning-for-dou-di-zhu","title":"Combinational Q-Learning for Dou Di Zhu","date":"2019-01-24","arxiv_id":"1901.08925","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-multi-step-deep-reinforcement","slug":"understanding-multi-step-deep-reinforcement","title":"Understanding Multi-Step Deep Reinforcement Learning: A Systematic Study of the DQN Target","date":"2019-01-22","arxiv_id":"1901.07510","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-recurrent-q-network-towards-self","slug":"a-deep-recurrent-q-network-towards-self","title":"A Deep Recurrent Q Network towards Self-adapting Distributed Microservices architecture","date":"2019-01-13","arxiv_id":"1901.04011","repositories_listed":1,"syntology":null},{"url":"/paper/information-directed-exploration-for-deep","slug":"information-directed-exploration-for-deep","title":"Information-Directed Exploration for Deep Reinforcement Learning","date":"2018-12-18","arxiv_id":"1812.07544","repositories_listed":1,"syntology":null},{"url":"/paper/urban-driving-with-multi-objective-deep","slug":"urban-driving-with-multi-objective-deep","title":"Urban Driving with Multi-Objective Deep Reinforcement Learning","date":"2018-11-21","arxiv_id":"1811.08586","repositories_listed":1,"syntology":null},{"url":"/paper/switch-based-active-deep-dyna-q-efficient","slug":"switch-based-active-deep-dyna-q-efficient","title":"Switch-based Active Deep Dyna-Q: Efficient Adaptive Planning for Task-Completion Dialogue Policy Learning","date":"2018-11-19","arxiv_id":"1811.07550","repositories_listed":1,"syntology":null},{"url":"/paper/deep-q-learning-for-fooling-neural-networks","slug":"deep-q-learning-for-fooling-neural-networks","title":"Deep Q learning for fooling neural networks","date":"2018-11-13","arxiv_id":"1811.05521","repositories_listed":1,"syntology":null},{"url":"/paper/double-q-pid-algorithm-for-mobile-robot","slug":"double-q-pid-algorithm-for-mobile-robot","title":"Double Q-PID algorithm for mobile robot control","date":"2018-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/actor-expert-a-framework-for-using-action","slug":"actor-expert-a-framework-for-using-action","title":"Greedy Actor-Critic: A New Conditional Cross-Entropy Method for Policy Improvement","date":"2018-10-22","arxiv_id":"1810.09103","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/actor-expert-a-framework-for-using-action#ran","syntology_url":"https://syntology.ai/paper/1810.09103","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.09103"}},"official":{"repos":["samuelfneumann/greedyac"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/assessing-the-potential-of-classical-q","slug":"assessing-the-potential-of-classical-q","title":"Assessing the Potential of Classical Q-learning in General Game Playing","date":"2018-10-14","arxiv_id":"1810.06078","repositories_listed":1,"syntology":null},{"url":"/paper/model-free-adaptive-optimal-control-of","slug":"model-free-adaptive-optimal-control-of","title":"Model-Free Adaptive Optimal Control of Episodic Fixed-Horizon Manufacturing Processes using Reinforcement Learning","date":"2018-09-18","arxiv_id":"1809.06646","repositories_listed":1,"syntology":null},{"url":"/paper/deterministic-implementations-for","slug":"deterministic-implementations-for","title":"Deterministic Implementations for Reproducibility in Deep Reinforcement Learning","date":"2018-09-15","arxiv_id":"1809.05676","repositories_listed":1,"syntology":null},{"url":"/paper/towards-better-interpretability-in-deep-q","slug":"towards-better-interpretability-in-deep-q","title":"Towards Better Interpretability in Deep Q-Networks","date":"2018-09-15","arxiv_id":"1809.05630","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-better-interpretability-in-deep-q#ran","syntology_url":"https://syntology.ai/paper/1809.05630","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.05630"}},"official":null}},{"url":"/paper/negative-update-intervals-in-deep-multi-agent","slug":"negative-update-intervals-in-deep-multi-agent","title":"Negative Update Intervals in Deep Multi-Agent Reinforcement Learning","date":"2018-09-13","arxiv_id":"1809.05096","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-deep-reinforcement-learning-for-1","slug":"multi-agent-deep-reinforcement-learning-for-1","title":"Multi-Agent Deep Reinforcement Learning for Dynamic Power Allocation in Wireless Networks","date":"2018-08-01","arxiv_id":"1808.00490","repositories_listed":1,"syntology":null},{"url":"/paper/is-q-learning-provably-efficient","slug":"is-q-learning-provably-efficient","title":"Is Q-learning Provably Efficient?","date":"2018-07-10","arxiv_id":"1807.03765","repositories_listed":1,"syntology":null},{"url":"/paper/using-reward-machines-for-high-level-task","slug":"using-reward-machines-for-high-level-task","title":"Using Reward Machines for High-Level Task Specification and Decomposition in Reinforcement Learning","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/gan-q-learning","slug":"gan-q-learning","title":"GAN Q-learning","date":"2018-05-13","arxiv_id":"1805.04874","repositories_listed":1,"syntology":null},{"url":"/paper/towards-symbolic-reinforcement-learning-with","slug":"towards-symbolic-reinforcement-learning-with","title":"Towards Symbolic Reinforcement Learning with Common Sense","date":"2018-04-23","arxiv_id":"1804.08597","repositories_listed":1,"syntology":null},{"url":"/paper/nonparametric-stochastic-compositional","slug":"nonparametric-stochastic-compositional","title":"Nonparametric Stochastic Compositional Gradient Descent for Q-Learning in Continuous Markov Decision Problems","date":"2018-04-19","arxiv_id":"1804.07323","repositories_listed":1,"syntology":null},{"url":"/paper/cytonrl-an-efficient-reinforcement-learning","slug":"cytonrl-an-efficient-reinforcement-learning","title":"CytonRL: an Efficient Reinforcement Learning Open-source Toolkit Implemented in C++","date":"2018-04-14","arxiv_id":"1804.05834","repositories_listed":1,"syntology":null},{"url":"/paper/composable-deep-reinforcement-learning-for","slug":"composable-deep-reinforcement-learning-for","title":"Composable Deep Reinforcement Learning for Robotic Manipulation","date":"2018-03-19","arxiv_id":"1803.06773","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-vision-based","slug":"deep-reinforcement-learning-for-vision-based","title":"Deep Reinforcement Learning for Vision-Based Robotic Grasping: A Simulated Comparative Evaluation of Off-Policy Methods","date":"2018-02-28","arxiv_id":"1802.10264","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-large-scale-fleet-management-via","slug":"efficient-large-scale-fleet-management-via","title":"Efficient Collaborative Multi-Agent Deep Reinforcement Learning for Large-Scale Fleet Management","date":"2018-02-18","arxiv_id":"1802.06444","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-q-learning-agent-for-the-l-game-with","slug":"a-deep-q-learning-agent-for-the-l-game-with","title":"A Deep Q-Learning Agent for the L-Game with Variable Batch Training","date":"2018-02-17","arxiv_id":"1802.06225","repositories_listed":1,"syntology":null},{"url":"/paper/qlbs-q-learner-in-the-black-scholes-merton","slug":"qlbs-q-learner-in-the-black-scholes-merton","title":"QLBS: Q-Learner in the Black-Scholes(-Merton) Worlds","date":"2017-12-13","arxiv_id":"1712.04609","repositories_listed":1,"syntology":null},{"url":"/paper/assumed-density-filtering-q-learning","slug":"assumed-density-filtering-q-learning","title":"Assumed Density Filtering Q-learning","date":"2017-12-09","arxiv_id":"1712.03333","repositories_listed":1,"syntology":null},{"url":"/paper/classification-with-costly-features-using","slug":"classification-with-costly-features-using","title":"Classification with Costly Features using Deep Reinforcement Learning","date":"2017-11-20","arxiv_id":"1711.07364","repositories_listed":1,"syntology":null},{"url":"/paper/the-effects-of-memory-replay-in-reinforcement","slug":"the-effects-of-memory-replay-in-reinforcement","title":"The Effects of Memory Replay in Reinforcement Learning","date":"2017-10-18","arxiv_id":"1710.06574","repositories_listed":1,"syntology":null},{"url":"/paper/automated-cloud-provisioning-on-aws-using","slug":"automated-cloud-provisioning-on-aws-using","title":"Automated Cloud Provisioning on AWS using Deep Reinforcement Learning","date":"2017-09-13","arxiv_id":"1709.04305","repositories_listed":1,"syntology":null},{"url":"/paper/practical-block-wise-neural-network","slug":"practical-block-wise-neural-network","title":"Practical Block-wise Neural Network Architecture Generation","date":"2017-08-18","arxiv_id":"1708.05552","repositories_listed":1,"syntology":null},{"url":"/paper/a-self-adaptive-proposal-model-for-temporal","slug":"a-self-adaptive-proposal-model-for-temporal","title":"A Self-Adaptive Proposal Model for Temporal Action Detection based on Reinforcement Learning","date":"2017-06-22","arxiv_id":"1706.07251","repositories_listed":1,"syntology":null},{"url":"/paper/generalized-value-iteration-networks-life","slug":"generalized-value-iteration-networks-life","title":"Generalized Value Iteration Networks: Life Beyond Lattices","date":"2017-06-08","arxiv_id":"1706.02416","repositories_listed":1,"syntology":null},{"url":"/paper/implications-of-decentralized-q-learning","slug":"implications-of-decentralized-q-learning","title":"Implications of Decentralized Q-learning Resource Allocation in Wireless Networks","date":"2017-05-30","arxiv_id":"1705.10508","repositories_listed":1,"syntology":null},{"url":"/paper/integral-policy-iterations-for-reinforcement","slug":"integral-policy-iterations-for-reinforcement","title":"Policy Iterations for Reinforcement Learning Problems in Continuous Time and Space -- Fundamental Theory and Methods","date":"2017-05-09","arxiv_id":"1705.03520","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-gap-between-value-and-policy","slug":"bridging-the-gap-between-value-and-policy","title":"Bridging the Gap Between Value and Policy Based Reinforcement Learning","date":"2017-02-28","arxiv_id":"1702.08892","repositories_listed":1,"syntology":null},{"url":"/paper/playing-doom-with-slam-augmented-deep","slug":"playing-doom-with-slam-augmented-deep","title":"Playing Doom with SLAM-Augmented Deep Reinforcement Learning","date":"2016-12-01","arxiv_id":"1612.00380","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-play-in-a-day-faster-deep","slug":"learning-to-play-in-a-day-faster-deep","title":"Learning to Play in a Day: Faster Deep Reinforcement Learning by Optimality Tightening","date":"2016-11-05","arxiv_id":"1611.01606","repositories_listed":1,"syntology":null},{"url":"/paper/active-exploration-in-parameterized","slug":"active-exploration-in-parameterized","title":"Active exploration in parameterized reinforcement learning","date":"2016-10-06","arxiv_id":"1610.01986","repositories_listed":1,"syntology":null},{"url":"/paper/angrier-birds-bayesian-reinforcement-learning","slug":"angrier-birds-bayesian-reinforcement-learning","title":"Angrier Birds: Bayesian reinforcement learning","date":"2016-01-06","arxiv_id":"1601.01297","repositories_listed":1,"syntology":null},{"url":"/paper/learning-simple-algorithms-from-examples","slug":"learning-simple-algorithms-from-examples","title":"Learning Simple Algorithms from Examples","date":"2015-11-23","arxiv_id":"1511.07275","repositories_listed":1,"syntology":null},{"url":"/paper/self-learning-cloud-controllers-fuzzy-q","slug":"self-learning-cloud-controllers-fuzzy-q","title":"Self-Learning Cloud Controllers: Fuzzy Q-Learning for Knowledge Evolution","date":"2015-07-02","arxiv_id":"1507.00567","repositories_listed":1,"syntology":null},{"url":"/paper/least-squares-policy-iteration","slug":"least-squares-policy-iteration","title":"Least-Squares Policy Iteration","date":"2003-12-04","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/between-mdps-and-semi-mdps-a-framework-for","slug":"between-mdps-and-semi-mdps-a-framework-for","title":"Between MDPs and semi-MDPs: A framework for temporal abstraction in reinforcement learning","date":"1999-08-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":null,"slug":"evaluating-reinforcement-learning-algorithms-1","title":"Evaluating Reinforcement Learning Algorithms for Navigation in Simulated Robotic Quadrupeds: A Comparative Study Inspired by Guide Dog Behaviour","date":"2025-07-17","arxiv_id":"2507.13277","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-data-ensemble-based-approach-for-sample","title":"A Data-Ensemble-Based Approach for Sample-Efficient LQ Control of Linear Time-Varying Systems","date":"2025-06-30","arxiv_id":"2506.23716","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-policy","title":"Reinforcement Learning-Based Policy Optimisation For Heterogeneous Radio Access","date":"2025-06-18","arxiv_id":"2506.15273","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-constraint-aware-off-policy","title":"Implicit Constraint-Aware Off-Policy Correction for Offline Reinforcement Learning","date":"2025-06-16","arxiv_id":"2506.14058","repositories_listed":0,"syntology":null},{"url":null,"slug":"reindsplit-reinforced-dynamic-split-learning","title":"ReinDSplit: Reinforced Dynamic Split Learning for Pest Recognition in Precision Agriculture","date":"2025-06-16","arxiv_id":"2506.13935","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-are-my-options-explaining-rl-agents-with","title":"\"What are my options?\": Explaining RL Agents with Diverse Near-Optimal Alternatives (Extended)","date":"2025-06-11","arxiv_id":"2506.09901","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-based-hierarchical-cooperative","title":"Q-learning-based Hierarchical Cooperative Local Search for Steelmaking-continuous Casting Scheduling Problem","date":"2025-06-10","arxiv_id":"2506.08608","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-optimal-q-learning-with-low-cost-for","title":"Regret-Optimal Q-Learning with Low Cost for Single-Agent and Federated Reinforcement Learning","date":"2025-06-05","arxiv_id":"2506.04626","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-performance-gap-between-target","title":"Bridging the Performance Gap Between Target-Free and Target-Based Reinforcement Learning With Iterated Q-Learning","date":"2025-06-04","arxiv_id":"2506.04398","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-performance-of-spike-based-deep-q","title":"Improving Performance of Spike-based Deep Q-Learning using Ternary Neurons","date":"2025-06-03","arxiv_id":"2506.03392","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-hanabi","title":"Reinforcement Learning for Hanabi","date":"2025-05-31","arxiv_id":"2506.00458","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropic-risk-optimization-in-discounted-mdps","title":"Entropic Risk Optimization in Discounted MDPs: Sample Complexity Bounds with a Generative Model","date":"2025-05-30","arxiv_id":"2506.00286","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-global-convergence-rates-for-federated","title":"On Global Convergence Rates for Federated Policy Gradient under Heterogeneous Environment","date":"2025-05-29","arxiv_id":"2505.23459","repositories_listed":0,"syntology":null},{"url":null,"slug":"boformer-learning-to-solve-multi-objective","title":"BOFormer: Learning to Solve Multi-Objective Bayesian Optimization via Non-Markovian RL","date":"2025-05-28","arxiv_id":"2505.21974","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-charge-more-a-theoretical-study","title":"Learning to Charge More: A Theoretical Study of Collusion by Q-Learning Agents","date":"2025-05-28","arxiv_id":"2505.22909","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-general-purpose-theorem-for-high","title":"A General-Purpose Theorem for High-Probability Bounds of Stochastic Approximation with Polyak Averaging","date":"2025-05-27","arxiv_id":"2505.21796","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-guarded-safe-reinforcement-learning","title":"Offline Guarded Safe Reinforcement Learning for Medical Treatment Optimization Strategies","date":"2025-05-22","arxiv_id":"2505.16242","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-stock-transactions","title":"Reinforcement Learning for Stock Transactions","date":"2025-05-22","arxiv_id":"2505.16099","repositories_listed":0,"syntology":null},{"url":null,"slug":"opa-pack-object-property-aware-robotic-bin","title":"OPA-Pack: Object-Property-Aware Robotic Bin Packing","date":"2025-05-19","arxiv_id":"2505.13339","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-a-reinforcement-learning-agent","title":"When a Reinforcement Learning Agent Encounters Unknown Unknowns","date":"2025-05-19","arxiv_id":"2505.13188","repositories_listed":0,"syntology":null},{"url":null,"slug":"imagination-limited-q-learning-for-offline","title":"Imagination-Limited Q-Learning for Offline Reinforcement Learning","date":"2025-05-18","arxiv_id":"2505.12211","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11081","title":"ShiQ: Bringing back Bellman to LLMs","date":"2025-05-16","arxiv_id":"2505.11081","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11478","title":"Automatic Reward Shaping from Confounded Offline Data","date":"2025-05-16","arxiv_id":"2505.11478","repositories_listed":0,"syntology":null},{"url":null,"slug":"bias-or-optimality-disentangling-bayesian","title":"Bias or Optimality? Disentangling Bayesian Inference and Learning Biases in Human Decision-Making","date":"2025-05-12","arxiv_id":"2505.08049","repositories_listed":0,"syntology":null},{"url":null,"slug":"convert-language-model-into-a-value-based","title":"Convert Language Model into a Value-based Strategic Planner","date":"2025-05-11","arxiv_id":"2505.06987","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-large-language-model-enhanced-q-learning","title":"A Large Language Model-Enhanced Q-learning for Capacitated Vehicle Routing Problem with Time Windows","date":"2025-05-09","arxiv_id":"2505.06178","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-approximation-theorem-for-deep-q","title":"Universal Approximation Theorem for Deep Q-Learning via FBSDE System","date":"2025-05-09","arxiv_id":"2505.06023","repositories_listed":0,"syntology":null},{"url":null,"slug":"merging-and-disentangling-views-in-visual","title":"Merging and Disentangling Views in Visual Reinforcement Learning for Robotic Manipulation","date":"2025-05-07","arxiv_id":"2505.04619","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlm-q-learning-aligning-vision-language","title":"VLM Q-Learning: Aligning Vision-Language Models for Interactive Decision-Making","date":"2025-05-06","arxiv_id":"2505.03181","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-approximation-theorem-of-deep-q","title":"Universal Approximation Theorem of Deep Q-Networks","date":"2025-05-04","arxiv_id":"2505.02288","repositories_listed":0,"syntology":null},{"url":null,"slug":"rank-one-modified-value-iteration","title":"Rank-One Modified Value Iteration","date":"2025-05-03","arxiv_id":"2505.01828","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-and-distributed-routing-in-iot","title":"Dynamic and Distributed Routing in IoT Networks based on Multi-Objective Q-Learning","date":"2025-05-01","arxiv_id":"2505.00918","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-neural-control-barrier-functions","title":"Learning Neural Control Barrier Functions from Offline Data with Conservatism","date":"2025-05-01","arxiv_id":"2505.00908","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-with-clustered-smart-csmart-data","title":"Q-Learning with Clustered-SMART (cSMART) Data: Examining Moderators in the Construction of Clustered Adaptive Interventions","date":"2025-05-01","arxiv_id":"2505.00822","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-double-deep-q-network-integrating","title":"Interactive Double Deep Q-network: Integrating Human Interventions and Evaluative Predictions in Reinforcement Learning of Autonomous Driving","date":"2025-04-28","arxiv_id":"2505.01440","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-asymptotic-guarantees-for-average-reward","title":"Non-Asymptotic Guarantees for Average-Reward Q-Learning with Adaptive Stepsizes","date":"2025-04-25","arxiv_id":"2504.18743","repositories_listed":0,"syntology":null},{"url":null,"slug":"sapo-rl-sequential-actuator-placement","title":"SAPO-RL: Sequential Actuator Placement Optimization for Fuselage Assembly via Reinforcement Learning","date":"2025-04-24","arxiv_id":"2504.17603","repositories_listed":0,"syntology":null}],"record_sha256":"05239d321643239f1b0ac7df76a7e784a926fa58df3f5cd570b04ced579f4ea6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}