{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/sequential-decision-making/papers/12","list_of":"/task/sequential-decision-making","task":"Sequential Decision Making","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":12,"pages_in_order":13,"rows_per_page":100,"rows":[1101,1200],"of":1210,"counts":{"archive_papers_tagged":1210,"with_a_code_link":351,"where_syntology_ran_a_sample":107,"not_listed_spam_title":0,"listed":1210,"listed_where_code_ran":107,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":90,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":90,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/sequential-decision-making","prev":"/task/sequential-decision-making/papers/11","next":"/task/sequential-decision-making/papers/13","papers":[{"url":null,"slug":"deep-reinforcement-learning-for-multi-agent","title":"Deep Reinforcement Learning for Multi-Agent Systems: A Review of Challenges, Solutions and Applications","date":"2018-12-31","arxiv_id":"1812.11794","repositories_listed":0,"syntology":null},{"url":null,"slug":"monte-carlo-tree-search-for-constrained","title":"Monte-Carlo Tree Search for Constrained POMDPs","date":"2018-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"negotiable-reinforcement-learning-for-pareto","title":"Negotiable Reinforcement Learning for Pareto Optimal Sequential Decision-Making","date":"2018-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-bounds-for-online-portfolio-selection","title":"Regret Bounds for Online Portfolio Selection with a Cardinality Constraint","date":"2018-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tight-bayesian-ambiguity-sets-for-robust-mdps","title":"Tight Bayesian Ambiguity Sets for Robust MDPs","date":"2018-11-15","arxiv_id":"1811.06512","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-for-multi-objective","title":"Meta-Learning for Multi-objective Reinforcement Learning","date":"2018-11-08","arxiv_id":"1811.03376","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-preserving-non-discrimination-when","title":"On preserving non-discrimination when combining expert advice","date":"2018-10-28","arxiv_id":"1810.11829","repositories_listed":0,"syntology":null},{"url":null,"slug":"resilient-computing-with-reinforcement","title":"Resilient Computing with Reinforcement Learning on a Dynamical System: Case Study in Sorting","date":"2018-09-25","arxiv_id":"1809.09261","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometric-multi-model-fitting-by-deep","title":"Geometric Multi-Model Fitting by Deep Reinforcement Learning","date":"2018-09-22","arxiv_id":"1809.08397","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-periodicity-with-temporal","title":"Predicting Periodicity with Temporal Difference Learning","date":"2018-09-20","arxiv_id":"1809.07435","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-convex-optimization-for-sequential","title":"Online Convex Optimization for Sequential Decision Processes and Extensive-Form Games","date":"2018-09-10","arxiv_id":"1809.03075","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-order-effects-in","title":"Investigating Order Effects in Multidimensional Relevance Judgment using Query Logs","date":"2018-07-14","arxiv_id":"1807.05355","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-summarisation-by-classification-with","title":"Video Summarisation by Classification with Deep Reinforcement Learning","date":"2018-07-09","arxiv_id":"1807.03089","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-against-nature-causal-discovery-for","title":"Playing against Nature: causal discovery for decision making under uncertainty","date":"2018-07-03","arxiv_id":"1807.01268","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-generative-adversarial-nets-with","title":"Multi-Task Generative Adversarial Nets with Shared Memory for Cross-Domain Coordination Control","date":"2018-07-01","arxiv_id":"1807.00298","repositories_listed":0,"syntology":null},{"url":null,"slug":"stagewise-safe-bayesian-optimization-with","title":"Stagewise Safe Bayesian Optimization with Gaussian Processes","date":"2018-06-20","arxiv_id":"1806.07555","repositories_listed":0,"syntology":null},{"url":null,"slug":"toprank-a-practical-algorithm-for-online","title":"TopRank: A practical algorithm for online stochastic ranking","date":"2018-06-06","arxiv_id":"1806.02248","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-self-imitating-diverse-policies","title":"Learning Self-Imitating Diverse Policies","date":"2018-05-25","arxiv_id":"1805.10309","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-online-exact-solutions-for-deterministic","title":"Fast Online Exact Solutions for Deterministic MDPs with Sparse Rewards","date":"2018-05-08","arxiv_id":"1805.02785","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-improving-deep-reinforcement-learning-for-1","title":"On Improving Deep Reinforcement Learning for POMDPs","date":"2018-04-17","arxiv_id":"1804.06309","repositories_listed":0,"syntology":null},{"url":null,"slug":"ucboost-a-boosting-approach-to-tame","title":"UCBoost: A Boosting Approach to Tame Complexity and Optimality for Stochastic Bandits","date":"2018-04-16","arxiv_id":"1804.05929","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-with-value-function","title":"Policy Gradient With Value Function Approximation For Collective Multiagent Planning","date":"2018-04-09","arxiv_id":"1804.02884","repositories_listed":0,"syntology":null},{"url":null,"slug":"hindsight-is-only-5050-unsuitability-of-mdp","title":"Hindsight is Only 50/50: Unsuitability of MDP based Approximate POMDP Solvers for Multi-resolution Information Gathering","date":"2018-04-07","arxiv_id":"1804.02573","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-e-commerce-search-engine-ranking","title":"Accelerating E-Commerce Search Engine Ranking by Contextual Factor Selection","date":"2018-03-14","arxiv_id":"1803.00693","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-imitation-and-reinforcement","title":"Hierarchical Imitation and Reinforcement Learning","date":"2018-03-01","arxiv_id":"1803.00590","repositories_listed":0,"syntology":null},{"url":null,"slug":"novel-approaches-to-accelerating-the","title":"Novel Approaches to Accelerating the Convergence Rate of Markov Decision Process for Search Result Diversification","date":"2018-02-23","arxiv_id":"1802.08401","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-anytime-algorithm-for-task-and-motion-mdps","title":"An Anytime Algorithm for Task and Motion MDPs","date":"2018-02-16","arxiv_id":"1802.05835","repositories_listed":0,"syntology":null},{"url":null,"slug":"mpc-inspired-neural-network-policies-for","title":"MPC-Inspired Neural Network Policies for Sequential Decision Making","date":"2018-02-15","arxiv_id":"1802.05803","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-human-behaviors-in-crowds-by","title":"Understanding Human Behaviors in Crowds by Imitating the Decision-Making Process","date":"2018-01-25","arxiv_id":"1801.08391","repositories_listed":0,"syntology":null},{"url":null,"slug":"testing-optimality-of-sequential-decision","title":"Testing Optimality of Sequential Decision-Making","date":"2018-01-04","arxiv_id":"1801.01574","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-shot-pedestrian-re-identification-via","title":"Multi-shot Pedestrian Re-identification via Sequential Decision Making","date":"2017-12-19","arxiv_id":"1712.07257","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-lda-uncovering-latent-patterns-in-text","title":"Q-LDA: Uncovering Latent Patterns in Text-based Sequential Decision Processes","date":"2017-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"loss-functions-for-multiset-prediction","title":"Loss Functions for Multiset Prediction","date":"2017-11-14","arxiv_id":"1711.05246","repositories_listed":0,"syntology":null},{"url":null,"slug":"servant-of-many-masters-shifting-priorities","title":"Servant of Many Masters: Shifting priorities in Pareto-optimal sequential decision-making","date":"2017-10-31","arxiv_id":"1711.00363","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-should-a-robot-assess-risk-towards-an","title":"How Should a Robot Assess Risk? Towards an Axiomatic Theory of Risk in Robotics","date":"2017-10-30","arxiv_id":"1710.11040","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-state-abstractions-for-decision","title":"Hierarchical State Abstractions for Decision-Making Problems with Computational Constraints","date":"2017-10-22","arxiv_id":"1710.07990","repositories_listed":0,"syntology":null},{"url":null,"slug":"asymmetric-actor-critic-for-image-based-robot","title":"Asymmetric Actor Critic for Image-Based Robot Learning","date":"2017-10-18","arxiv_id":"1710.06542","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-learning-for-sequential-decision","title":"Optimal Learning for Sequential Decision Making for Expensive Cost Functions with Stochastic Binary Feedbacks","date":"2017-09-13","arxiv_id":"1709.05216","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-aware-algorithms-for-adversarial","title":"Safety-Aware Algorithms for Adversarial Contextual Bandit","date":"2017-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"non-stationary-bandits-with-habituation-and","title":"Non-Stationary Bandits with Habituation and Recovery Dynamics","date":"2017-07-26","arxiv_id":"1707.08423","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-for-multi-robot-cooperation-in","title":"Learning for Multi-robot Cooperation in Partially Observable Stochastic Environments with Macro-actions","date":"2017-07-24","arxiv_id":"1707.07399","repositories_listed":0,"syntology":null},{"url":null,"slug":"correlational-dueling-bandits-with","title":"Correlational Dueling Bandits with Application to Clinical Treatment in Large Decision Spaces","date":"2017-07-08","arxiv_id":"1707.02375","repositories_listed":0,"syntology":null},{"url":null,"slug":"tableaux-for-policy-synthesis-for-mdps-with","title":"Tableaux for Policy Synthesis for MDPs with PCTL* Constraints","date":"2017-06-30","arxiv_id":"1706.10102","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-theory-is-predictive-but-is-it-complete","title":"The Theory is Predictive, but is it Complete? An Application to Human Perception of Randomness","date":"2017-06-21","arxiv_id":"1706.06974","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlocking-the-potential-of-simulators-design","title":"Unlocking the Potential of Simulators: Design with RL in Mind","date":"2017-06-08","arxiv_id":"1706.02501","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-method-for-the-online-construction-of-the","title":"A method for the online construction of the set of states of a Markov Decision Process using Answer Set Programming","date":"2017-06-05","arxiv_id":"1706.01417","repositories_listed":0,"syntology":null},{"url":null,"slug":"boltzmann-exploration-done-right","title":"Boltzmann Exploration Done Right","date":"2017-05-29","arxiv_id":"1705.10257","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-mix-n-step-returns-generalizing","title":"Learning to Mix n-Step Returns: Generalizing lambda-Returns for Deep Reinforcement Learning","date":"2017-05-21","arxiv_id":"1705.07445","repositories_listed":0,"syntology":null},{"url":null,"slug":"answer-set-programming-for-non-stationary","title":"Answer Set Programming for Non-Stationary Markov Decision Processes","date":"2017-05-03","arxiv_id":"1705.01399","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-reinforcement-learning-for-demand","title":"Using Reinforcement Learning for Demand Response of Domestic Hot Water Buffers: a Real-Life Demonstration","date":"2017-03-16","arxiv_id":"1703.05486","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimizing-maximum-regret-in-commitment","title":"Minimizing Maximum Regret in Commitment Constrained Sequential Decision Making","date":"2017-03-14","arxiv_id":"1703.04587","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-robust-kalman-filter","title":"Deep Robust Kalman Filter","date":"2017-03-07","arxiv_id":"1703.02310","repositories_listed":0,"syntology":null},{"url":null,"slug":"deeply-aggrevated-differentiable-imitation","title":"Deeply AggreVaTeD: Differentiable Imitation Learning for Sequential Prediction","date":"2017-03-03","arxiv_id":"1703.01030","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-learning-for-accurate-estimation-of","title":"Active Learning for Accurate Estimation of Linear Models","date":"2017-03-02","arxiv_id":"1703.00579","repositories_listed":0,"syntology":null},{"url":null,"slug":"tight-bounds-for-bandit-combinatorial","title":"Tight Bounds for Bandit Combinatorial Optimization","date":"2017-02-24","arxiv_id":"1702.07539","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-repeat-fine-grained-action","title":"Learning to Repeat: Fine Grained Action Repetition for Deep Reinforcement Learning","date":"2017-02-20","arxiv_id":"1702.06054","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-visual-object","title":"Deep Reinforcement Learning for Visual Object Tracking in Videos","date":"2017-01-31","arxiv_id":"1701.08936","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-control-of-thermostatically","title":"Model-Free Control of Thermostatically Controlled Loads Connected to a District Heating Network","date":"2017-01-27","arxiv_id":"1701.08074","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-contextual-bandit-approach-for-stream-based","title":"A Contextual Bandit Approach for Stream-Based Active Learning","date":"2017-01-24","arxiv_id":"1701.06725","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-negotiable-reinforcement-learning","title":"Toward negotiable reinforcement learning: shifting priorities in Pareto optimal sequential decision-making","date":"2017-01-05","arxiv_id":"1701.01302","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-planning-and-lifted-inference","title":"Stochastic Planning and Lifted Inference","date":"2017-01-04","arxiv_id":"1701.01048","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-preference-based-to-multiobjective","title":"From Preference-Based to Multiobjective Sequential Decision-Making","date":"2017-01-03","arxiv_id":"1701.00646","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-armed-bandits-competing-with-optimal","title":"Multi-armed Bandits: Competing with Optimal Sequences","date":"2016-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-video-classification-via-adaptive","title":"Fast Video Classification via Adaptive Cascading of Deep Models","date":"2016-11-20","arxiv_id":"1611.06453","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-problem-approximate-planning-of-pomdps","title":"Open Problem: Approximate Planning of POMDPs in the class of Memoryless Policies","date":"2016-08-17","arxiv_id":"1608.04996","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-collective-intelligence-as-distributed","title":"Human collective intelligence as distributed Bayesian inference","date":"2016-08-05","arxiv_id":"1608.01987","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-policy-improvement-by-minimizing-robust","title":"Safe Policy Improvement by Minimizing Robust Baseline Regret","date":"2016-07-13","arxiv_id":"1607.03842","repositories_listed":0,"syntology":null},{"url":null,"slug":"preference-at-first-sight","title":"Preference at First Sight","date":"2016-06-24","arxiv_id":"1606.07524","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-bayesian-linear-information-filtering","title":"The Bayesian Linear Information Filtering Problem","date":"2016-05-30","arxiv_id":"1605.09088","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-action-sequence-learning-for-causal","title":"Deep Action Sequence Learning for Causal Shape Transformation","date":"2016-05-17","arxiv_id":"1605.05368","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-web-scale-event-summarization-using","title":"Real-Time Web Scale Event Summarization Using Sequential Decision Making","date":"2016-05-12","arxiv_id":"1605.03664","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-contextual-bandits-with-known","title":"Stochastic Contextual Bandits with Known Reward Functions","date":"2016-04-30","arxiv_id":"1605.00176","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-for-reward-design-to-improve","title":"Deep Learning for Reward Design to Improve Monte Carlo Tree Search in ATARI Games","date":"2016-04-24","arxiv_id":"1604.07095","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-sensing-via-multi-armed-bandit","title":"Optimal Sensing via Multi-armed Bandit Relaxations in Mixed Observability Domains","date":"2016-03-15","arxiv_id":"1603.04586","repositories_listed":0,"syntology":null},{"url":null,"slug":"pac-reinforcement-learning-with-rich","title":"PAC Reinforcement Learning with Rich Observations","date":"2016-02-08","arxiv_id":"1602.02722","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-constrained-reinforcement-learning-with","title":"Risk-Constrained Reinforcement Learning with Percentile Risk Criteria","date":"2015-12-05","arxiv_id":"1512.01629","repositories_listed":0,"syntology":null},{"url":null,"slug":"reuse-of-neural-modules-for-general-video","title":"Reuse of Neural Modules for General Video Game Playing","date":"2015-12-04","arxiv_id":"1512.01537","repositories_listed":0,"syntology":null},{"url":null,"slug":"bandits-with-unobserved-confounders-a-causal","title":"Bandits with Unobserved Confounders: A Causal Approach","date":"2015-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-applied-to-an-electric","title":"Reinforcement Learning Applied to an Electric Water Heater: From Theory to Practice","date":"2015-11-29","arxiv_id":"1512.00408","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-transition-independent-multi-agent","title":"Solving Transition-Independent Multi-agent MDPs with Sparse Interactions (Extended version)","date":"2015-11-29","arxiv_id":"1511.09047","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-knowledge-gradient-with-logistic-belief","title":"The Knowledge Gradient with Logistic Belief Models for Binary Classification","date":"2015-10-08","arxiv_id":"1510.02354","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-phase-q-learning-for-bidding-based","title":"Two Phase $Q-$learning for Bidding-based Vehicle Sharing","date":"2015-09-29","arxiv_id":"1509.08932","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimization-of-anemia-treatment-in","title":"Optimization of anemia treatment in hemodialysis patients via reinforcement learning","date":"2015-09-14","arxiv_id":"1509.03977","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-efficient-representations-for","title":"Learning Efficient Representations for Reinforcement Learning","date":"2015-08-28","arxiv_id":"1509.02413","repositories_listed":0,"syntology":null},{"url":null,"slug":"experimental-analysis-of-data-driven-control","title":"Experimental analysis of data-driven control for a building heating system","date":"2015-07-13","arxiv_id":"1507.03638","repositories_listed":0,"syntology":null},{"url":null,"slug":"utility-based-dueling-bandits-as-a-partial","title":"Utility-based Dueling Bandits as a Partial Monitoring Game","date":"2015-07-10","arxiv_id":"1507.02750","repositories_listed":0,"syntology":null},{"url":null,"slug":"hands-on-learning-to-search-for-structured","title":"Hands-on Learning to Search for Structured Prediction","date":"2015-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"global-bandits","title":"Global Bandits","date":"2015-03-29","arxiv_id":"1503.08370","repositories_listed":0,"syntology":null},{"url":null,"slug":"second-order-quantile-methods-for-experts-and","title":"Second-order Quantile Methods for Experts and Combinatorial Games","date":"2015-02-27","arxiv_id":"1502.08009","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairness-in-multi-agent-sequential-decision","title":"Fairness in Multi-Agent Sequential Decision-Making","date":"2014-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"active-sensing-as-bayes-optimal-sequential","title":"Active Sensing as Bayes-Optimal Sequential Decision Making","date":"2014-08-09","arxiv_id":"1408.2056","repositories_listed":0,"syntology":null},{"url":null,"slug":"chasing-ghosts-competing-with-stateful","title":"Chasing Ghosts: Competing with Stateful Policies","date":"2014-07-29","arxiv_id":"1407.7635","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithms-for-cvar-optimization-in-mdps","title":"Algorithms for CVaR Optimization in MDPs","date":"2014-06-12","arxiv_id":"1406.3339","repositories_listed":0,"syntology":null},{"url":null,"slug":"proximal-reinforcement-learning-a-new-theory","title":"Proximal Reinforcement Learning: A New Theory of Sequential Decision Making in Primal-Dual Spaces","date":"2014-05-26","arxiv_id":"1405.6757","repositories_listed":0,"syntology":null},{"url":null,"slug":"variance-constrained-actor-critic-algorithms","title":"Variance-Constrained Actor-Critic Algorithms for Discounted and Average Reward MDPs","date":"2014-03-25","arxiv_id":"1403.6530","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-multi-objective-sequential","title":"A Survey of Multi-Objective Sequential Decision-Making","date":"2014-02-04","arxiv_id":"1402.0590","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-model-equivalences-for-solving","title":"Exploiting Model Equivalences for Solving Interactive Dynamic Influence Diagrams","date":"2014-01-18","arxiv_id":"1401.4600","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-deterministic-policies-in-markovian","title":"Non-Deterministic Policies in Markovian Decision Processes","date":"2014-01-16","arxiv_id":"1401.3871","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-planning-algorithms-for-pomdps","title":"Online Planning Algorithms for POMDPs","date":"2014-01-15","arxiv_id":"1401.3436","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-algorithms-for-risk-sensitive","title":"Actor-Critic Algorithms for Risk-Sensitive MDPs","date":"2013-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"4e902cbb71699de6c23fc4bd405f212e7e333f310876eeeb6facec278532ebf2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}