{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/sequential-decision-making/papers/11","list_of":"/task/sequential-decision-making","task":"Sequential Decision Making","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":13,"rows_per_page":100,"rows":[1001,1100],"of":1210,"counts":{"archive_papers_tagged":1210,"with_a_code_link":351,"where_syntology_ran_a_sample":107,"not_listed_spam_title":0,"listed":1210,"listed_where_code_ran":107,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":90,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":90,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/sequential-decision-making","prev":"/task/sequential-decision-making/papers/10","next":"/task/sequential-decision-making/papers/12","papers":[{"url":null,"slug":"circuit-routing-using-monte-carlo-tree-search","title":"Circuit Routing Using Monte Carlo Tree Search and Deep Neural Networks","date":"2020-06-24","arxiv_id":"2006.13607","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-reinforcement-learning-a-1","title":"Risk-Sensitive Reinforcement Learning: a Martingale Approach to Reward Uncertainty","date":"2020-06-23","arxiv_id":"2006.12686","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-optimism-in-model-based-reinforcement","title":"Towards Tractable Optimism in Model-Based Reinforcement Learning","date":"2020-06-21","arxiv_id":"2006.11911","repositories_listed":0,"syntology":null},{"url":null,"slug":"counterfactually-guided-policy-transfer-in","title":"Counterfactually Guided Off-policy Transfer in Clinical Settings","date":"2020-06-20","arxiv_id":"2006.11654","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-by-repetition-stochastic-multi-armed","title":"Learning by Repetition: Stochastic Multi-armed Bandits under Priming Effect","date":"2020-06-18","arxiv_id":"2006.10356","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameterized-mdps-and-reinforcement-learning","title":"Parameterized MDPs and Reinforcement Learning Problems -- A Maximum Entropy Principle Based Framework","date":"2020-06-17","arxiv_id":"2006.09646","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-relationship-between-structure-in","title":"On the Relationship Between Structure in Natural Language and Models of Sequential Decision Processes","date":"2020-06-12","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"group-fair-online-allocation-in-continuous","title":"Group-Fair Online Allocation in Continuous Time","date":"2020-06-11","arxiv_id":"2006.06852","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-human-driving-behavior-through","title":"Modeling Human Driving Behavior through Generative Adversarial Imitation Learning","date":"2020-06-10","arxiv_id":"2006.06412","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-is-particle-filtering-efficient-for","title":"When is Particle Filtering Efficient for Planning in Partially Observed Linear Dynamical Systems?","date":"2020-06-10","arxiv_id":"2006.05975","repositories_listed":0,"syntology":null},{"url":null,"slug":"stealing-deep-reinforcement-learning-models","title":"Stealing Deep Reinforcement Learning Models for Fun and Profit","date":"2020-06-09","arxiv_id":"2006.05032","repositories_listed":0,"syntology":null},{"url":null,"slug":"sharp-thresholds-of-the-information-cascade","title":"Sharp Thresholds of the Information Cascade Fragility Under a Mismatched Model","date":"2020-06-07","arxiv_id":"2006.04117","repositories_listed":0,"syntology":null},{"url":null,"slug":"global-convergence-of-maml-for-lqr","title":"When Does MAML Objective Have Benign Landscape?","date":"2020-05-31","arxiv_id":"2006.00453","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-bi-objective-routing-of-multiple","title":"Dynamic Bi-Objective Routing of Multiple Vehicles","date":"2020-05-28","arxiv_id":"2005.13872","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-measure-reinforcement-learning-for","title":"Active Measure Reinforcement Learning for Observation Cost Minimization","date":"2020-05-26","arxiv_id":"2005.12697","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-bayesian-optimization","title":"Causal Bayesian Optimization","date":"2020-05-24","arxiv_id":"2005.11741","repositories_listed":0,"syntology":null},{"url":null,"slug":"implementability-of-honest-multi-agent","title":"Implementability of Honest Multi-Agent Sequential Decision-Making with Dynamic Population","date":"2020-05-19","arxiv_id":"2003.03173","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-first-order-methods-for-robust-mdps","title":"Scalable First-Order Methods for Robust MDPs","date":"2020-05-11","arxiv_id":"2005.05434","repositories_listed":0,"syntology":null},{"url":null,"slug":"divide-and-conquer-monte-carlo-tree-search","title":"Divide-and-Conquer Monte Carlo Tree Search For Goal-Directed Planning","date":"2020-04-23","arxiv_id":"2004.11410","repositories_listed":0,"syntology":null},{"url":null,"slug":"icorpp-interleaved-commonsense-reasoning-and","title":"iCORPP: Interleaved Commonsense Reasoning and Probabilistic Planning on Robots","date":"2020-04-18","arxiv_id":"2004.08672","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-deep-reinforcement-learning-for-1","title":"Actor-Critic Deep Reinforcement Learning for Solving Job Shop Scheduling Problems","date":"2020-04-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-batch-learning-in-finite-action","title":"Sequential Batch Learning in Finite-Action Linear Contextual Bandits","date":"2020-04-14","arxiv_id":"2004.06321","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-framework-for-2","title":"A Deep Reinforcement Learning Framework for Continuous Intraday Market Bidding","date":"2020-04-13","arxiv_id":"2004.05940","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-learning-sequential-decision","title":"Distributed Learning: Sequential Decision Making in Resource-Constrained Environments","date":"2020-04-13","arxiv_id":"2004.06171","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-ai-interaction-loop-training-new","title":"Human AI interaction loop training: New approach for interactive reinforcement learning","date":"2020-03-09","arxiv_id":"2003.04203","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-farewell-to-arms-sequential-reward","title":"A Farewell to Arms: Sequential Reward Maximization on a Budget with a Giving Up Option","date":"2020-03-06","arxiv_id":"2003.03456","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-robustness-and-regularization","title":"Distributional Robustness and Regularization in Reinforcement Learning","date":"2020-03-05","arxiv_id":"2003.02894","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-exploitation-in-constrained-mdps","title":"Exploration-Exploitation in Constrained MDPs","date":"2020-03-04","arxiv_id":"2003.02189","repositories_listed":0,"syntology":null},{"url":null,"slug":"structure-adaptive-sequential-testing-for","title":"Structure-Adaptive Sequential Testing for Online False Discovery Rate Control","date":"2020-02-28","arxiv_id":"2003.00113","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-directed-sampling-for-linear","title":"Information Directed Sampling for Linear Partial Monitoring","date":"2020-02-25","arxiv_id":"2002.11182","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-batch-decision-making-with-high","title":"Online Batch Decision-Making with High-Dimensional Covariates","date":"2020-02-21","arxiv_id":"2002.09438","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-multi-output-regression-via","title":"Weakly-supervised Multi-output Regression via Correlated Gaussian Processes","date":"2020-02-19","arxiv_id":"2002.08412","repositories_listed":0,"syntology":null},{"url":null,"slug":"legion-best-first-concolic-testing","title":"Legion: Best-First Concolic Testing","date":"2020-02-15","arxiv_id":"2002.06311","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-generalization-of-reinforcement","title":"Improving Generalization of Reinforcement Learning with Minimax Distributional Soft Actor-Critic","date":"2020-02-13","arxiv_id":"2002.05502","repositories_listed":0,"syntology":null},{"url":null,"slug":"listwise-learning-to-rank-with-deep-q","title":"Listwise Learning to Rank with Deep Q-Networks","date":"2020-02-13","arxiv_id":"2002.07651","repositories_listed":0,"syntology":null},{"url":null,"slug":"tight-lower-bounds-for-combinatorial-multi","title":"Tight Lower Bounds for Combinatorial Multi-Armed Bandits","date":"2020-02-13","arxiv_id":"2002.05392","repositories_listed":0,"syntology":null},{"url":null,"slug":"verifiable-rnn-based-policies-for-pomdps","title":"Verifiable RNN-Based Policies for POMDPs Under Temporal Logic Constraints","date":"2020-02-13","arxiv_id":"2002.05615","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-reinforcement-learning-for","title":"Accelerating Reinforcement Learning for Reaching using Continuous Curriculum Learning","date":"2020-02-07","arxiv_id":"2002.02697","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-providing-post-hoc-symbolic","title":"Bridging the Gap: Providing Post-Hoc Symbolic Explanations for Sequential Decision-Making Problems with Inscrutable Representations","date":"2020-02-04","arxiv_id":"2002.01080","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairness-in-learning-based-sequential","title":"Fairness in Learning-Based Sequential Decision Algorithms: A Survey","date":"2020-01-14","arxiv_id":"2001.04861","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-sequential-resource-investment-planning","title":"A storage expansion planning framework using reinforcement learning and simulation-based optimization","date":"2020-01-10","arxiv_id":"2001.03507","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-computation-and-generalization-of-1","title":"On Computation and Generalization of Generative Adversarial Imitation Learning","date":"2020-01-09","arxiv_id":"2001.02792","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-and-indirect-reinforcement-learning","title":"Direct and indirect reinforcement learning","date":"2019-12-23","arxiv_id":"1912.10600","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-multi-agent-reinforcement","title":"Decentralized Multi-Agent Reinforcement Learning with Networked Agents: Recent Advances","date":"2019-12-09","arxiv_id":"1912.03821","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-entropy-monte-carlo-planning","title":"Maximum Entropy Monte-Carlo Planning","date":"2019-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-optimized-and-energy-efficient-parallel","title":"An Optimized and Energy-Efficient Parallel Implementation of Non-Iteratively Trained Recurrent Neural Networks","date":"2019-11-26","arxiv_id":"1911.13252","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-a","title":"Multi-Agent Reinforcement Learning: A Selective Overview of Theories and Algorithms","date":"2019-11-24","arxiv_id":"1911.10635","repositories_listed":0,"syntology":null},{"url":null,"slug":"working-memory-graphs","title":"Working Memory Graphs","date":"2019-11-17","arxiv_id":"1911.07141","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-shot-learning-and-behavioral-eligibility","title":"One-shot learning and behavioral eligibility traces in sequential decision making","date":"2019-11-12","arxiv_id":"1707.04192","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptivity-in-adaptive-submodularity","title":"Adaptivity in Adaptive Submodularity","date":"2019-11-09","arxiv_id":"1911.03620","repositories_listed":0,"syntology":null},{"url":null,"slug":"beta-dvbf-learning-state-space-models-for","title":"Beta DVBF: Learning State-Space Models for Control from High Dimensional Observations","date":"2019-11-02","arxiv_id":"1911.00756","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-exploration-in-linear-contextual","title":"Adaptive Exploration in Linear Contextual Bandit","date":"2019-10-15","arxiv_id":"1910.06996","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-multi-objective","title":"Reinforcement Learning for Multi-Objective Optimization of Online Decisions in High-Dimensional Systems","date":"2019-10-01","arxiv_id":"1910.00211","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-choice-function-framework-for-online","title":"The Choice Function Framework for Online Policy Improvement","date":"2019-10-01","arxiv_id":"1910.00614","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-inter-agent-knowledge","title":"Collaborative Inter-agent Knowledge Distillation for Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizing-reinforcement-learning-to-unseen","title":"Generalizing Reinforcement Learning to Unseen Actions","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-functionally-decomposed-hierarchies","title":"Learning Functionally Decomposed Hierarchies for Continuous Navigation Tasks","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-benefits-of-deep-hierarchical-rl","title":"PROVABLY BENEFITS OF DEEP HIERARCHICAL RL","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-task-driven","title":"Selective Network Discovery via Deep Reinforcement Learning on Embedded Spaces","date":"2019-09-16","arxiv_id":"1909.07294","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-arm-wise-randomization-approach-to","title":"An Arm-Wise Randomization Approach to Combinatorial Linear Semi-Bandits","date":"2019-09-05","arxiv_id":"1909.02251","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-a-user-guess-what-her-followers-want","title":"Can A User Anticipate What Her Followers Want?","date":"2019-09-01","arxiv_id":"1909.00440","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-healthcare-a-survey","title":"Reinforcement Learning in Healthcare: A Survey","date":"2019-08-22","arxiv_id":"1908.08796","repositories_listed":0,"syntology":null},{"url":null,"slug":"190807808","title":"Exploring Offline Policy Evaluation for the Continuous-Armed Bandit Problem","date":"2019-08-21","arxiv_id":"1908.07808","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-planning-for-decentralized-stochastic","title":"Online Planning for Decentralized Stochastic Control with Partial History Sharing","date":"2019-08-06","arxiv_id":"1908.02357","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-commonsense-reasoning-and","title":"Bridging Commonsense Reasoning and Probabilistic Planning via a Probabilistic Action Language","date":"2019-07-31","arxiv_id":"1907.13482","repositories_listed":0,"syntology":null},{"url":null,"slug":"bandit-convex-optimization-in-non-stationary","title":"Bandit Convex Optimization in Non-stationary Environments","date":"2019-07-29","arxiv_id":"1907.12340","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-discovery-of-decision-states-for","title":"IR-VIC: Unsupervised Discovery of Sub-goals for Transfer in RL","date":"2019-07-24","arxiv_id":"1907.10580","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-sufficient-statistic-for-influence-in","title":"A Sufficient Statistic for Influence in Structured Multiagent Environments","date":"2019-07-22","arxiv_id":"1907.09278","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-advancement-transforming-policy-under","title":"Reward Advancement: Transforming Policy under Maximum Causal Entropy Principle","date":"2019-07-11","arxiv_id":"1907.05390","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-scheme-for-dynamic-risk-sensitive","title":"A Scheme for Dynamic Risk-Sensitive Sequential Decision Making","date":"2019-07-09","arxiv_id":"1907.04269","repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-on-symmetric-stable-bandits","title":"Thompson Sampling on Symmetric $α$-Stable Bandits","date":"2019-07-08","arxiv_id":"1907.03821","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-relevance-for-online-decision","title":"Exploiting Relevance for Online Decision-Making in High-Dimensions","date":"2019-07-01","arxiv_id":"1907.00783","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-markov-models-via-low-rank","title":"Learning Markov models via low-rank optimization","date":"2019-06-28","arxiv_id":"1907.00113","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-theoretical-connection-between-statistical","title":"A Theoretical Connection Between Statistical Physics and Reinforcement Learning","date":"2019-06-24","arxiv_id":"1906.10228","repositories_listed":0,"syntology":null},{"url":null,"slug":"macro-action-multi-timescale-dynamic","title":"Macro-action Multi-time scale Dynamic Programming for Energy Management in Buildings with Phase Change Materials","date":"2019-06-11","arxiv_id":"1906.05200","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-heterogeneous-scheduler","title":"Neural Heterogeneous Scheduler","date":"2019-06-09","arxiv_id":"1906.03724","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-under-drift","title":"Non-Stationary Reinforcement Learning: The Blessing of (More) Optimism","date":"2019-06-07","arxiv_id":"1906.02922","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-and-transferable-learning-of","title":"Learning NP-Hard Multi-Agent Assignment Planning using GNN: Inference on a Random Graph and Provable Auction-Fitted Q-learning","date":"2019-05-29","arxiv_id":"1905.12204","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-based-sequential-decision-making","title":"Knowledge-Based Sequential Decision-Making Under Uncertainty","date":"2019-05-16","arxiv_id":"1905.07030","repositories_listed":0,"syntology":null},{"url":null,"slug":"tight-regret-bounds-for-infinite-armed-linear","title":"Tight Regret Bounds for Infinite-armed Linear Contextual Bandits","date":"2019-05-04","arxiv_id":"1905.01435","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-term-impact-of-fair-machine-learning-in","title":"Group Retention when Using Machine Learning in Sequential Decision Making: the Interplay between User Dynamics and Fairness","date":"2019-05-02","arxiv_id":"1905.00569","repositories_listed":0,"syntology":null},{"url":null,"slug":"soft-q-learning-with-mutual-information","title":"Soft Q-Learning with Mutual-Information Regularization","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"trajectory-vae-for-multi-modal-imitation","title":"Trajectory VAE for multi-modal imitation","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-generalizing-alphago-zero","title":"Understanding & Generalizing AlphaGo Zero","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-optimal-1","title":"Deep Reinforcement Learning for Optimal Critical Care Pain Management with Morphine using Dueling Double-Deep Q Networks","date":"2019-04-25","arxiv_id":"1904.11115","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-adaptive-submodularity-approximation","title":"Beyond Adaptive Submodularity: Approximation Guarantees of Greedy Policy with Adaptive Submodularity Ratio","date":"2019-04-24","arxiv_id":"1904.10748","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-variable-algorithms-for-multimodal","title":"Latent Variable Algorithms for Multimodal Learning and Sensor Fusion","date":"2019-04-23","arxiv_id":"1904.10450","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-short-survey-on-memory-based-reinforcement","title":"A Short Survey On Memory Based Reinforcement Learning","date":"2019-04-14","arxiv_id":"1904.06736","repositories_listed":0,"syntology":null},{"url":null,"slug":"similarities-between-policy-gradient-methods","title":"Similarities between policy gradient methods (PGM) in Reinforcement learning (RL) and supervised learning (SL)","date":"2019-04-12","arxiv_id":"1904.06260","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-surrogate-models-for-sequential","title":"Meta-Learning surrogate models for sequential decision making","date":"2019-03-28","arxiv_id":"1903.11907","repositories_listed":0,"syntology":null},{"url":null,"slug":"automating-predictive-modeling-process-using","title":"Automating Predictive Modeling Process using Reinforcement Learning","date":"2019-03-02","arxiv_id":"1903.00743","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-of-intentional-backdoors-in-sequential","title":"Design of intentional backdoors in sequential models","date":"2019-02-26","arxiv_id":"1902.09972","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-in-the-loop-active-covariance-learning","title":"Human-in-the-loop Active Covariance Learning for Improving Prediction in Small Data Sets","date":"2019-02-26","arxiv_id":"1902.09834","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-thompson-sampling-via-optimal","title":"Scalable Thompson Sampling via Optimal Transport","date":"2019-02-19","arxiv_id":"1902.07239","repositories_listed":0,"syntology":null},{"url":null,"slug":"network-offloading-policies-for-cloud","title":"Network Offloading Policies for Cloud Robotics: a Learning-based Approach","date":"2019-02-15","arxiv_id":"1902.05703","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-the-discount-factor-in","title":"Rethinking the Discount Factor in Reinforcement Learning: A Decision Theoretic Approach","date":"2019-02-08","arxiv_id":"1902.02893","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyper-parameter-tuning-under-a-budget","title":"Hyper-parameter Tuning under a Budget Constraint","date":"2019-02-01","arxiv_id":"1902.00532","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-neural-linear-bandits-overcoming","title":"Deep Neural Linear Bandits: Overcoming Catastrophic Forgetting through Likelihood Matching","date":"2019-01-24","arxiv_id":"1901.08612","repositories_listed":0,"syntology":null},{"url":null,"slug":"robot-sequential-decision-making-using-lstm","title":"Learning and Reasoning for Robot Sequential Decision Making under Uncertainty","date":"2019-01-16","arxiv_id":"1901.05322","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-computational-framework-for-motor-skill","title":"A Computational Framework for Motor Skill Acquisition","date":"2019-01-03","arxiv_id":"1901.01856","repositories_listed":0,"syntology":null}],"record_sha256":"7882cdd03eb0deff00cd8c1679aa88f4b87b7f7dfb15a18d8da28c76f86651da","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}