{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/131","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":131,"pages_in_order":132,"rows_per_page":100,"rows":[13001,13100],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/130","next":"/task/reinforcement-learning/papers/132","papers":[{"url":null,"slug":"thompson-sampling-for-learning-parameterized","title":"Thompson Sampling for Learning Parameterized Markov Decision Processes","date":"2014-06-29","arxiv_id":"1406.7498","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-and-imitation-learning-via","title":"Reinforcement and Imitation Learning via Interactive No-Regret Learning","date":"2014-06-23","arxiv_id":"1406.5979","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-medical-treatments-using-novel","title":"Personalized Medical Treatments Using Novel Reinforcement Learning Algorithms","date":"2014-06-16","arxiv_id":"1406.3922","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-reinforcement-learning-with","title":"Multi-objective Reinforcement Learning with Continuous Pareto Frontier Approximation Supplementary Material","date":"2014-06-13","arxiv_id":"1406.3497","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-and-the","title":"Model-based Reinforcement Learning and the Eluder Dimension","date":"2014-06-07","arxiv_id":"1406.1853","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-multi-label-classification-with","title":"Comparing Multi-label Classification with Reinforcement Learning for Summarisation of Time-series Data","date":"2014-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"single-agent-vs-multi-agent-techniques-for","title":"Single-Agent vs. Multi-Agent Techniques for Concurrent Reinforcement Learning of Negotiation Dialogue Policies","date":"2014-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"proximal-reinforcement-learning-a-new-theory","title":"Proximal Reinforcement Learning: A New Theory of Sequential Decision Making in Primal-Dual Spaces","date":"2014-05-26","arxiv_id":"1405.6757","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-model-learning-for-human-robot","title":"Efficient Model Learning for Human-Robot Collaborative Tasks","date":"2014-05-24","arxiv_id":"1405.6341","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-shaping-ensembles-in-reinforcement","title":"Off-Policy Shaping Ensembles in Reinforcement Learning","date":"2014-05-21","arxiv_id":"1405.5358","repositories_listed":0,"syntology":null},{"url":null,"slug":"projective-simulation-applied-to-the-grid","title":"Projective simulation applied to the grid-world and the mountain-car problem","date":"2014-05-21","arxiv_id":"1405.5459","repositories_listed":0,"syntology":null},{"url":null,"slug":"selecting-near-optimal-approximate-state","title":"Selecting Near-Optimal Approximate State Representations in Reinforcement Learning","date":"2014-05-12","arxiv_id":"1405.2652","repositories_listed":0,"syntology":null},{"url":null,"slug":"structural-return-maximization-for","title":"Structural Return Maximization for Reinforcement Learning","date":"2014-05-12","arxiv_id":"1405.2606","repositories_listed":0,"syntology":null},{"url":null,"slug":"dinasti-dialogues-with-a-negotiating","title":"DINASTI: Dialogues with a Negotiating Appointment Setting Interface","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"optimistic-risk-perception-in-the-temporal","title":"Optimistic Risk Perception in the Temporal Difference error Explains the Relation between Risk-taking, Gambling, Sensation-seeking and Low Fear","date":"2014-04-08","arxiv_id":"1404.2078","repositories_listed":0,"syntology":null},{"url":null,"slug":"undirected-machine-translation-with","title":"Undirected Machine Translation with Discriminative Reinforcement Learning","date":"2014-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-cooperative-cognitive-relaying-and","title":"Optimal Cooperative Cognitive Relaying and Spectrum Access for an Energy Harvesting Cognitive Radio: Reinforcement Learning Approach","date":"2014-03-30","arxiv_id":"1403.7735","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-multi-agent-and-single-agent","title":"Comparison of Multi-agent and Single-agent Inverse Learning on a Simulated Soccer Example","date":"2014-03-26","arxiv_id":"1403.6822","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-inverse-reinforcement-learning","title":"Multi-agent Inverse Reinforcement Learning for Two-person Zero-sum Games","date":"2014-03-25","arxiv_id":"1403.6508","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-mcmc-based-inference-in","title":"Adaptive MCMC-Based Inference in Probabilistic Logic Programs","date":"2014-03-24","arxiv_id":"1403.6036","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-perturbation-algorithms-for","title":"Simultaneous Perturbation Algorithms for Batch Off-Policy Search","date":"2014-03-18","arxiv_id":"1403.4514","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-reinforcement-learning-in","title":"Near-optimal Reinforcement Learning in Factored MDPs","date":"2014-03-15","arxiv_id":"1403.3741","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsically-motivated-learning-of-visual","title":"Intrinsically Motivated Learning of Visual Motion Perception and Smooth Pursuit","date":"2014-02-14","arxiv_id":"1402.3344","repositories_listed":0,"syntology":null},{"url":null,"slug":"better-optimism-by-bayes-adaptive-planning","title":"Better Optimism By Bayes: Adaptive Planning with Rich Models","date":"2014-02-09","arxiv_id":"1402.1958","repositories_listed":0,"syntology":null},{"url":null,"slug":"frequency-based-patrolling-with-heterogeneous","title":"Frequency-Based Patrolling with Heterogeneous Agents and Limited Communication","date":"2014-02-07","arxiv_id":"1402.1757","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-stochastic-optimization-under","title":"Online Stochastic Optimization under Correlated Bandit Feedback","date":"2014-02-04","arxiv_id":"1402.0562","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-exploration-of-state-and-action-spaces","title":"Safe Exploration of State and Action Spaces in Reinforcement Learning","date":"2014-02-04","arxiv_id":"1402.0560","repositories_listed":0,"syntology":null},{"url":null,"slug":"kalman-temporal-differences","title":"Kalman Temporal Differences","date":"2014-01-16","arxiv_id":"1406.3270","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-deterministic-policies-in-markovian","title":"Non-Deterministic Policies in Markovian Decision Processes","date":"2014-01-16","arxiv_id":"1401.3871","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multiagent-reinforcement-learning-algorithm","title":"A Multiagent Reinforcement Learning Algorithm with Non-linear Dynamics","date":"2014-01-15","arxiv_id":"1401.3454","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-partially-observable-deterministic","title":"Learning Partially Observable Deterministic Action Models","date":"2014-01-15","arxiv_id":"1401.3437","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-generalisation-symmetries-in","title":"Exploiting generalisation symmetries in accuracy-based learning classifier systems: An initial study","date":"2014-01-10","arxiv_id":"1401.2949","repositories_listed":0,"syntology":null},{"url":null,"slug":"dj-mc-a-reinforcement-learning-agent-for","title":"DJ-MC: A Reinforcement-Learning Agent for Music Playlist Recommendation","date":"2014-01-09","arxiv_id":"1401.1880","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-demand-response-using-device-based","title":"Optimal Demand Response Using Device Based Reinforcement Learning","date":"2014-01-08","arxiv_id":"1401.1549","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-policy-evaluation-under-multiple","title":"Distributed Policy Evaluation Under Multiple Behavior Strategies","date":"2013-12-30","arxiv_id":"1312.7606","repositories_listed":0,"syntology":null},{"url":null,"slug":"avoiding-confusion-between-predictors-and","title":"Avoiding Confusion between Predictors and Inhibitors in Value Function Approximation","date":"2013-12-19","arxiv_id":"1312.5714","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-step-size-for-policy-gradient","title":"Adaptive Step-Size for Policy Gradient Methods","date":"2013-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bellman-error-based-feature-generation-using","title":"Bellman Error Based Feature Generation using Random Projections on Sparse Spaces","date":"2013-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-exploration-and-value-function","title":"Efficient Exploration and Value Function Generalization in Deterministic Systems","date":"2013-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-learning-and-planning-with","title":"Efficient Learning and Planning with Compressed Predictive States","date":"2013-12-01","arxiv_id":"1312.0286","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimistic-policy-iteration-and-natural-actor","title":"Optimistic policy iteration and natural actor-critic: A unifying view and a non-optimality result","date":"2013-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-shaping-integrating-human-feedback","title":"Policy Shaping: Integrating Human Feedback with Reinforcement Learning","date":"2013-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"projected-natural-actor-critic","title":"Projected Natural Actor-Critic","date":"2013-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-robust-markov","title":"Reinforcement Learning in Robust Markov Decision Processes","date":"2013-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-reinforcement-learning-for-h_infty","title":"Off-policy reinforcement learning for $ H_\\infty $ control design","date":"2013-11-24","arxiv_id":"1311.6107","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-definition-of-a-general-learning","title":"On the definition of a general learning system with user-defined operators","date":"2013-11-18","arxiv_id":"1311.4235","repositories_listed":0,"syntology":null},{"url":null,"slug":"clustering-markov-decision-processes-for","title":"Clustering Markov Decision Processes For Continual Transfer","date":"2013-11-15","arxiv_id":"1311.3959","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-reinforcement-learning","title":"Risk-sensitive Reinforcement Learning","date":"2013-11-08","arxiv_id":"1311.2097","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-deep-and-recurrent-architectures","title":"Exploring Deep and Recurrent Architectures for Optimal Control","date":"2013-11-07","arxiv_id":"1311.1761","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-in-interactive-personalized-music","title":"Exploration in Interactive Personalized Music Recommendation: A Reinforcement Learning Approach","date":"2013-11-06","arxiv_id":"1311.6355","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-based-approximate-policy-iteration-for","title":"Data-based approximate policy iteration for nonlinear continuous-time optimal control design","date":"2013-11-02","arxiv_id":"1311.0396","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-matrix","title":"Reinforcement Learning for Matrix Computations: PageRank as an Example","date":"2013-11-01","arxiv_id":"1311.2889","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-framework-for","title":"Reinforcement Learning Framework for Opportunistic Routing in WSNs","date":"2013-10-31","arxiv_id":"1310.8467","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-reinforcement-learning-via-gossip","title":"Distributed Reinforcement Learning via Gossip","date":"2013-10-28","arxiv_id":"1310.7610","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-complexity-of-multi-task-reinforcement","title":"Sample Complexity of Multi-task Reinforcement Learning","date":"2013-09-26","arxiv_id":"1309.6821","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-difference-learning-to-assist-human","title":"Temporal-Difference Learning to Assist Human Decision Making during the Control of an Artificial Limb","date":"2013-09-18","arxiv_id":"1309.4714","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-sample-complexity-of-general","title":"The Sample-Complexity of General Reinforcement Learning","date":"2013-08-22","arxiv_id":"1308.4828","repositories_listed":0,"syntology":null},{"url":null,"slug":"coevolutionary-networks-of-reinforcement","title":"Coevolutionary networks of reinforcement-learning agents","date":"2013-08-05","arxiv_id":"1308.1049","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-state-representations-for","title":"Evaluating State Representations for Reinforcement Learning of Turn-Taking Policies in Tutorial Dialogue","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-student-feedback-from-time-series","title":"Generating Student Feedback from Time-Series Data Using Reinforcement Learning","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-two-issue","title":"Reinforcement Learning of Two-Issue Negotiation Dialogue Policies","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-transfer-in-multi-armed-bandit","title":"Sequential Transfer in Multi-armed Bandit with Finite Set of Models","date":"2013-07-25","arxiv_id":"1307.6887","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-policy-gradients-with-parameter","title":"Model-Based Policy Gradients with Parameter-Based Exploration by Least-Squares Conditional Density Estimation","date":"2013-07-19","arxiv_id":"1307.5118","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-in","title":"Efficient Reinforcement Learning in Deterministic Systems with Value Function Generalization","date":"2013-07-18","arxiv_id":"1307.4847","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-inverse-reinforcement-learning-1","title":"Probabilistic inverse reinforcement learning in unknown environments","date":"2013-07-14","arxiv_id":"1307.3785","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-policy-search","title":"Multi-Task Policy Search","date":"2013-07-02","arxiv_id":"1307.0813","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-up-robust-mdps-by-reinforcement","title":"Scaling Up Robust MDPs by Reinforcement Learning","date":"2013-06-26","arxiv_id":"1306.6189","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-restrictions-on","title":"Reinforcement learning with restrictions on the action set","date":"2013-06-12","arxiv_id":"1306.2918","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-association-problem-in-wireless-networks","title":"The association problem in wireless networks: a Policy Gradient Reinforcement Learning approach","date":"2013-06-11","arxiv_id":"1306.2554","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-uncertainty-estimation-in","title":"Direct Uncertainty Estimation in Reinforcement Learning","date":"2013-06-06","arxiv_id":"1306.1553","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-search-any-local-optimum-enjoys-a","title":"Policy Search: Any Local Optimum Enjoys a Global Performance Guarantee","date":"2013-06-06","arxiv_id":"1306.1520","repositories_listed":0,"syntology":null},{"url":null,"slug":"more-efficient-reinforcement-learning-via","title":"(More) Efficient Reinforcement Learning via Posterior Sampling","date":"2013-06-04","arxiv_id":"1306.0940","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-transfer-of-evolved-artificial","title":"Real-world Transfer of Evolved Artificial Immune System Behaviours between Small and Large Scale Robotic Platforms","date":"2013-05-31","arxiv_id":"1305.7432","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-the-soccer","title":"Reinforcement Learning for the Soccer Dribbling Task","date":"2013-05-28","arxiv_id":"1305.6568","repositories_listed":0,"syntology":null},{"url":null,"slug":"estimating-or-propagating-gradients-through","title":"Estimating or Propagating Gradients Through Stochastic Neurons","date":"2013-05-14","arxiv_id":"1305.2982","repositories_listed":0,"syntology":null},{"url":null,"slug":"geiringer-theorems-from-population-genetics","title":"Geiringer Theorems: From Population Genetics to Computational Intelligence, Memory Evolutive Systems and Hebbian Learning","date":"2013-05-11","arxiv_id":"1305.2504","repositories_listed":0,"syntology":null},{"url":null,"slug":"cover-tree-bayesian-reinforcement-learning","title":"Cover Tree Bayesian Reinforcement Learning","date":"2013-05-08","arxiv_id":"1305.1809","repositories_listed":0,"syntology":null},{"url":null,"slug":"projective-simulation-for-classical-learning","title":"Projective simulation for classical learning agents: a comprehensive investigation","date":"2013-05-07","arxiv_id":"1305.1578","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-bounds-for-reinforcement-learning-with","title":"Regret Bounds for Reinforcement Learning with Policy Advice","date":"2013-05-05","arxiv_id":"1305.1027","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-deterministic-logic-programs","title":"Non Deterministic Logic Programs","date":"2013-04-26","arxiv_id":"1304.7168","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-general-framework-for-interacting-bayes","title":"A General Framework for Interacting Bayes-Optimally with Self-Interested Agents using Arbitrary Parametric Model and Model Prior","date":"2013-04-07","arxiv_id":"1304.2024","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-bayesian-reinforcement-learning","title":"Model-based Bayesian Reinforcement Learning for Dialogue Management","date":"2013-04-05","arxiv_id":"1304.1819","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-for-a-darwinian-brain-part-2-cognitive","title":"Design for a Darwinian Brain: Part 2. Cognitive Architecture","date":"2013-03-28","arxiv_id":"1303.7201","repositories_listed":0,"syntology":null},{"url":null,"slug":"abc-reinforcement-learning","title":"ABC Reinforcement Learning","date":"2013-03-27","arxiv_id":"1303.6977","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-for-high","title":"Efficient Reinforcement Learning for High Dimensional Linear Quadratic Systems","date":"2013-03-24","arxiv_id":"1303.5984","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-greedy-approximation-of-bayesian","title":"A Greedy Approximation of Bayesian Reinforcement Learning with Probably Optimistic Transition Model","date":"2013-03-13","arxiv_id":"1303.3163","repositories_listed":0,"syntology":null},{"url":null,"slug":"toggling-a-genetic-switch-using-reinforcement","title":"Toggling a Genetic Switch Using Reinforcement Learning","date":"2013-03-12","arxiv_id":"1303.3183","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-q-learning-applied-to-ubiquitous","title":"Hybrid Q-Learning Applied to Ubiquitous recommender system","date":"2013-03-10","arxiv_id":"1303.2651","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-regret-bounds-for-undiscounted","title":"Online Regret Bounds for Undiscounted Continuous Reinforcement Learning","date":"2013-02-11","arxiv_id":"1302.2550","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-optimal-reward-baseline-for-gradient","title":"The Optimal Reward Baseline for Gradient-Based Reinforcement Learning","date":"2013-01-10","arxiv_id":"1301.2315","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-port-hamiltonian","title":"Reinforcement learning for port-Hamiltonian systems","date":"2012-12-21","arxiv_id":"1212.5524","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithms-for-learning-markov-field-policies","title":"Algorithms for Learning Markov Field Policies","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-hierarchical-reinforcement-learning","title":"Bayesian Hierarchical Reinforcement Learning","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-sensitive-exploration-in-bayesian","title":"Cost-Sensitive Exploration in Bayesian Reinforcement Learning","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-in-model-based-reinforcement","title":"Exploration in Model-based Reinforcement Learning by Empirically Estimating Learning Progress","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-optimistic-region-selection","title":"Hierarchical Optimistic Region Selection driven by Curiosity","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-through","title":"Inverse Reinforcement Learning through Structured Classification","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learned-prioritization-for-trading-off","title":"Learned Prioritization for Trading Off Accuracy and Speed","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neurally-plausible-reinforcement-learning-of","title":"Neurally Plausible Reinforcement Learning of Working Memory Tasks","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nonparametric-bayesian-inverse-reinforcement","title":"Nonparametric Bayesian Inverse Reinforcement Learning for Multiple Reward Functions","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"762fb6765a0dc0a20773de662806bc79e5cde94fac99c26d72417b9c6b5e821f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}