{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/sequential-decision-making/papers/10","list_of":"/task/sequential-decision-making","task":"Sequential Decision Making","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":13,"rows_per_page":100,"rows":[901,1000],"of":1210,"counts":{"archive_papers_tagged":1210,"with_a_code_link":351,"where_syntology_ran_a_sample":107,"not_listed_spam_title":0,"listed":1210,"listed_where_code_ran":107,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":90,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":90,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/sequential-decision-making","prev":"/task/sequential-decision-making/papers/9","next":"/task/sequential-decision-making/papers/11","papers":[{"url":null,"slug":"not-all-users-are-the-same-providing","title":"Not all users are the same: Providing personalized explanations for sequential decision making problems","date":"2021-06-23","arxiv_id":"2106.12207","repositories_listed":0,"syntology":null},{"url":null,"slug":"lorenz-system-state-stability-identification","title":"Lorenz System State Stability Identification using Neural Networks","date":"2021-06-16","arxiv_id":"2106.08489","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-dag-search","title":"Probabilistic DAG Search","date":"2021-06-16","arxiv_id":"2106.08717","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-modular-framework-for-object-based-saccadic","title":"A modular framework for object-based saccadic decisions in dynamic scenes","date":"2021-06-10","arxiv_id":"2106.06073","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-avoidance-and-overvaluation-in","title":"Information Avoidance and Overvaluation in Sequential Decision Making under Epistemic Constraints","date":"2021-06-09","arxiv_id":"2106.04984","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-reliable-priors-in-the-function","title":"Meta-Learning Reliable Priors in the Function Space","date":"2021-06-06","arxiv_id":"2106.03195","repositories_listed":0,"syntology":null},{"url":null,"slug":"heuristic-guided-reinforcement-learning","title":"Heuristic-Guided Reinforcement Learning","date":"2021-06-05","arxiv_id":"2106.02757","repositories_listed":0,"syntology":null},{"url":null,"slug":"be-considerate-objectives-side-effects-and","title":"Be Considerate: Objectives, Side Effects, and Deciding How to Act","date":"2021-06-04","arxiv_id":"2106.02617","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-schedule-job-shop-problems","title":"Learning to schedule job-shop problems: Representation and policy learning using graph neural network and reinforcement learning","date":"2021-06-02","arxiv_id":"2106.01086","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-mdps-from-features-predict-then-1","title":"Learning MDPs from Features: Predict-Then-Optimize for Sequential Decision Making by Reinforcement Learning","date":"2021-05-21","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"techniques-toward-optimizing-viewability-in","title":"Techniques Toward Optimizing Viewability in RTB Ad Campaigns Using Reinforcement Learning","date":"2021-05-21","arxiv_id":"2105.10587","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-control-of-robust-team-stochastic","title":"Robust optimal policies for team Markov games","date":"2021-05-16","arxiv_id":"2105.07405","repositories_listed":0,"syntology":null},{"url":null,"slug":"bandit-based-centralized-matching-in-two","title":"Bandit based centralized matching in two-sided markets for peer to peer lending","date":"2021-05-06","arxiv_id":"2105.02589","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-reinforcement-learning-for-1","title":"Data-Efficient Reinforcement Learning for Malaria Control","date":"2021-05-04","arxiv_id":"2105.01620","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-and-cumulative-constraint-violation","title":"Regret and Cumulative Constraint Violation Analysis for Distributed Online Constrained Convex Optimization","date":"2021-05-01","arxiv_id":"2105.00321","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistical-inference-with-m-estimators-on","title":"Statistical Inference with M-Estimators on Adaptively Collected Data","date":"2021-04-29","arxiv_id":"2104.14074","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-using-guided","title":"Reinforcement Learning using Guided Observability","date":"2021-04-22","arxiv_id":"2104.10986","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-an-aid-policy-to-minimize-student","title":"Discovering an Aid Policy to Minimize Student Evasion Using Offline Reinforcement Learning","date":"2021-04-20","arxiv_id":"2104.10258","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-comfort-aware-reinforcement-learning","title":"Visual Comfort Aware-Reinforcement Learning for Depth Adjustment of Stereoscopic 3D Images","date":"2021-04-14","arxiv_id":"2104.06782","repositories_listed":0,"syntology":null},{"url":null,"slug":"jamming-resilient-path-planning-for-multiple","title":"Jamming-Resilient Path Planning for Multiple UAVs via Deep Reinforcement Learning","date":"2021-04-09","arxiv_id":"2104.04477","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-uav-trajectory-optimization","title":"Learning-Based UAV Trajectory Optimization with Collision Avoidance and Connectivity Constraints","date":"2021-04-03","arxiv_id":"2104.06256","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-convex-optimization-with-continuous","title":"Online Convex Optimization with Continuous Switching Constraint","date":"2021-03-21","arxiv_id":"2103.11370","repositories_listed":0,"syntology":null},{"url":null,"slug":"forward-and-backward-bellman-equations","title":"Forward and Backward Bellman equations improve the efficiency of EM algorithm for DEC-POMDP","date":"2021-03-19","arxiv_id":"2103.10752","repositories_listed":0,"syntology":null},{"url":null,"slug":"situated-language-learning-via-interactive","title":"Situated Language Learning via Interactive Narratives","date":"2021-03-18","arxiv_id":"2103.09977","repositories_listed":0,"syntology":null},{"url":null,"slug":"homomorphically-encrypted-linear-contextual","title":"Encrypted Linear Contextual Bandit","date":"2021-03-17","arxiv_id":"2103.09927","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-for-planning-automatic","title":"Meta-Learning for Planning: Automatic Synthesis of Sample Based Planners","date":"2021-03-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-sequential-decision-making-with","title":"Optimal sequential decision making with probabilistic digital twins","date":"2021-03-12","arxiv_id":"2103.07405","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-goal-generation-using-dynamical-1","title":"Automatic Goal Generation using Dynamical Distance Learning","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bandit-linear-optimization-for-sequential","title":"Bandit Linear Optimization for Sequential Decision Making and Extensive-Form Games","date":"2021-03-08","arxiv_id":"2103.04546","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-online-learning-in-unknown","title":"Model-Free Online Learning in Unknown Sequential Decision Making Problems and Games","date":"2021-03-08","arxiv_id":"2103.04539","repositories_listed":0,"syntology":null},{"url":null,"slug":"batched-neural-bandits","title":"Batched Neural Bandits","date":"2021-02-25","arxiv_id":"2102.13028","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperparameter-transfer-learning-with","title":"Hyperparameter Transfer Learning with Adaptive Complexity","date":"2021-02-25","arxiv_id":"2102.12810","repositories_listed":0,"syntology":null},{"url":null,"slug":"sentinel-taming-uncertainty-with-ensemble","title":"SENTINEL: Taming Uncertainty with Ensemble-based Distributional Reinforcement Learning","date":"2021-02-22","arxiv_id":"2102.11075","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-meta-reinforcement-learning-using","title":"Model-based Meta Reinforcement Learning using Graph Structured Surrogate Models","date":"2021-02-16","arxiv_id":"2102.08291","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-markov-decision-processes-learning","title":"Causal Markov Decision Processes: Learning Good Interventions Efficiently","date":"2021-02-15","arxiv_id":"2102.07663","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-portfolio-1","title":"Deep Reinforcement Learning for Portfolio Optimization using Latent Feature State Space (LFSS) Module","date":"2021-02-11","arxiv_id":"2102.06233","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-matters-offline-pretraining","title":"Representation Matters: Offline Pretraining for Sequential Decision Making","date":"2021-02-11","arxiv_id":"2102.05815","repositories_listed":0,"syntology":null},{"url":null,"slug":"patterns-predictions-and-actions-a-story","title":"Patterns, predictions, and actions: A story about machine learning","date":"2021-02-10","arxiv_id":"2102.05242","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-frame-skipping-in","title":"An Analysis of Frame-skipping in Reinforcement Learning","date":"2021-02-07","arxiv_id":"2102.03718","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-modularized-and-scalable-multi-agent","title":"MSPM: A Modularized and Scalable Multi-Agent Reinforcement Learning-based System for Financial Portfolio Management","date":"2021-02-06","arxiv_id":"2102.03502","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-human-decision-making-by","title":"Improving Human Decision-Making by Discovering Efficient Strategies for Hierarchical Planning","date":"2021-01-31","arxiv_id":"2102.00521","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-machine-learning-help-in-solving-cargo","title":"Reinforcement Learning for Freight Booking Control Problems","date":"2021-01-29","arxiv_id":"2102.00092","repositories_listed":0,"syntology":null},{"url":null,"slug":"coordiq-coordinated-q-learning-for-electric","title":"CoordiQ : Coordinated Q-learning for Electric Vehicle Charging Recommendation","date":"2021-01-28","arxiv_id":"2102.00847","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-minerl-2020-competition-on-sample","title":"The MineRL 2020 Competition on Sample Efficient Reinforcement Learning using Human Priors","date":"2021-01-26","arxiv_id":"2101.11071","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-confidence-off-policy-or-counterfactual","title":"High-Confidence Off-Policy (or Counterfactual) Variance Estimation","date":"2021-01-25","arxiv_id":"2101.09847","repositories_listed":0,"syntology":null},{"url":null,"slug":"gst-group-sparse-training-for-accelerating","title":"GST: Group-Sparse Training for Accelerating Deep Reinforcement Learning","date":"2021-01-24","arxiv_id":"2101.09650","repositories_listed":0,"syntology":null},{"url":null,"slug":"deciding-what-to-learn-a-rate-distortion","title":"Deciding What to Learn: A Rate-Distortion Approach","date":"2021-01-15","arxiv_id":"2101.06197","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-imitative-graph-representation","title":"Reinforced Imitative Graph Representation Learning for Mobile User Profiling: An Adversarial Training Perspective","date":"2021-01-07","arxiv_id":"2101.02634","repositories_listed":0,"syntology":null},{"url":null,"slug":"computing-preimages-of-deep-neural-networks","title":"Computing Preimages of Deep Neural Networks with Applications to Safety","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"divide-and-conquer-monte-carlo-tree-search-1","title":"Divide-and-Conquer Monte Carlo Tree Search","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-make-decisions-via-submodular","title":"Learning to Make Decisions via Submodular Regularization","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-recover-from-failures-using","title":"Learning to Recover from Failures using Memory","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-and-leveraging-causal-relations","title":"Understanding and Leveraging Causal Relations in Deep Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-constrained-policies-via-mutual","title":"Privacy-Constrained Policies via Mutual Information Regularized Policy Gradients","date":"2020-12-30","arxiv_id":"2012.15019","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-regret-bound-for-non-stationary-multi-armed","title":"A Regret bound for Non-stationary Multi-Armed Bandits with Fairness Constraints","date":"2020-12-24","arxiv_id":"2012.13380","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-charging-of-electric-vehicle","title":"Autonomous Charging of Electric Vehicle Fleets to Enhance Renewable Generation Dispatchability","date":"2020-12-22","arxiv_id":"2012.12257","repositories_listed":0,"syntology":null},{"url":null,"slug":"demystify-painting-with-rl","title":"Demystify Painting with RL","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-mobile-robot-navigation-in-the-dense","title":"Learning Mobile Robot Navigation in the Dense Crowd with Deep Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"delay-and-cooperation-in-nonstochastic-linear","title":"Delay and Cooperation in Nonstochastic Linear Bandits","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-online-rent-or-buy-algorithms-with","title":"Improving Online Rent-or-Buy Algorithms with Sequential Decision Making and ML Predictions","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-policy-gradient-primal-dual-method","title":"Natural Policy Gradient Primal-Dual Method for Constrained Markov Decision Processes","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-efficiency-in-hierarchical-reinforcement","title":"On Efficiency in Hierarchical Reinforcement Learning","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-with-general-objective-functions","title":"Planning with General Objective Functions: Going Beyond Total Rewards","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"r-learning-in-actor-critic-model-offers-a","title":"R-learning in actor-critic model offers a biologically relevant mechanism for sequential decision-making","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"modality-buffet-for-real-time-object","title":"Modality-Buffet for Real-Time Object Detection","date":"2020-11-17","arxiv_id":"2011.08726","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-batch-policy-learning-in-markov","title":"Robust Batch Policy Learning in Markov Decision Processes","date":"2020-11-09","arxiv_id":"2011.04185","repositories_listed":0,"syntology":null},{"url":null,"slug":"reliable-off-policy-evaluation-for","title":"Reliable Off-policy Evaluation for Reinforcement Learning","date":"2020-11-08","arxiv_id":"2011.04102","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-and-multi-agent-deep-reinforcement","title":"Single and Multi-Agent Deep Reinforcement Learning for AI-Enabled Wireless Networks: A Tutorial","date":"2020-11-06","arxiv_id":"2011.03615","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-multiple-intelligent-reflecting","title":"Multi-IRS-assisted Multi-Cell Uplink MIMO Communications under Imperfect CSI: A Deep Reinforcement Learning Approach","date":"2020-11-02","arxiv_id":"2011.01141","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-efficient-active","title":"Reinforcement Learning with Efficient Active Feature Acquisition","date":"2020-11-02","arxiv_id":"2011.00825","repositories_listed":0,"syntology":null},{"url":null,"slug":"bandits-in-matching-markets-ideas-and","title":"Bandits in Matching Markets: Ideas and Proposals for Peer Lending","date":"2020-10-30","arxiv_id":"2011.04400","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-are-the-statistical-limits-of-offline-rl","title":"What are the Statistical Limits of Offline RL with Linear Function Approximation?","date":"2020-10-22","arxiv_id":"2010.11895","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-network-based-adaptive-alert-threshold","title":"Deep Q-Network-based Adaptive Alert Threshold Selection Policy for Payment Fraud Systems in Retail Banking","date":"2020-10-21","arxiv_id":"2010.11062","repositories_listed":0,"syntology":null},{"url":null,"slug":"dba-bandits-self-driving-index-tuning-under","title":"DBA bandits: Self-driving index tuning under ad-hoc, analytical workloads with safety guarantees","date":"2020-10-19","arxiv_id":"2010.09208","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-generative-machine-learning-approach-to","title":"A Generative Machine Learning Approach to Policy Optimization in Pursuit-Evasion Games","date":"2020-10-04","arxiv_id":"2010.01711","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-with-expected-quadratic-1","title":"Mean-Variance Efficient Reinforcement Learning with Applications to Dynamic Financial Investment","date":"2020-10-03","arxiv_id":"2010.01404","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-reinforcement-learning-more-difficult-than","title":"Is Reinforcement Learning More Difficult Than Bandits? A Near-optimal Algorithm Escaping the Curse of Horizon","date":"2020-09-28","arxiv_id":"2009.13503","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-sample-efficient-algorithm-for-episodic","title":"A Sample-Efficient Algorithm for Episodic Finite-Horizon MDP with Constraints","date":"2020-09-23","arxiv_id":"2009.11348","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-discovery-for-causal-bandits-utilizing","title":"Causal Bandits without prior knowledge using separating sets","date":"2020-09-16","arxiv_id":"2009.07916","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-in-deep-reinforcement","title":"Transfer Learning in Deep Reinforcement Learning: A Survey","date":"2020-09-16","arxiv_id":"2009.07888","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-the-fundamental-limits-of-imitation","title":"Toward the Fundamental Limits of Imitation Learning","date":"2020-09-13","arxiv_id":"2009.05990","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-inspection-and-maintenance-planning","title":"Optimal Inspection and Maintenance Planning for Deteriorating Structural Components through Dynamic Bayesian Networks and Markov Decision Processes","date":"2020-09-09","arxiv_id":"2009.04547","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-policy-evaluation-for-value-based","title":"Inverse Policy Evaluation for Value-based Sequential Decision-making","date":"2020-08-26","arxiv_id":"2008.11329","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-privacy-pricing-the-interplay-between","title":"Spatial Privacy Pricing: The Interplay between Privacy, Utility and Price in Geo-Marketplaces","date":"2020-08-25","arxiv_id":"2008.11817","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-knowledge-based-sequential","title":"A Survey of Knowledge-based Sequential Decision Making under Uncertainty","date":"2020-08-19","arxiv_id":"2008.08548","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-deep-reinforcement-learning-for-1","title":"Deep Model-Based Reinforcement Learning for High-Dimensional Problems, a Survey","date":"2020-08-11","arxiv_id":"2008.05598","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-machine-of-few-words-interactive-speaker","title":"A Machine of Few Words -- Interactive Speaker Recognition with Reinforcement Learning","date":"2020-08-07","arxiv_id":"2008.03127","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamics-generalization-via-information","title":"Dynamics Generalization via Information Bottleneck in Deep Reinforcement Learning","date":"2020-08-03","arxiv_id":"2008.00614","repositories_listed":0,"syntology":null},{"url":null,"slug":"tracking-the-race-between-deep-reinforcement","title":"Tracking the Race Between Deep Reinforcement Learning and Imitation Learning -- Extended Version","date":"2020-08-03","arxiv_id":"2008.00766","repositories_listed":0,"syntology":null},{"url":null,"slug":"compare-and-select-video-summarization-with","title":"Compare and Select: Video Summarization with Multi-Agent Reinforcement Learning","date":"2020-07-29","arxiv_id":"2007.14552","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-visuomotor-policy-training","title":"Data-efficient visuomotor policy training using reinforcement learning and generative models","date":"2020-07-26","arxiv_id":"2007.13134","repositories_listed":0,"syntology":null},{"url":null,"slug":"aircaprl-autonomous-aerial-human-motion","title":"AirCapRL: Autonomous Aerial Human Motion Capture using Deep Reinforcement Learning","date":"2020-07-13","arxiv_id":"2007.06343","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-based-rl-agents-with-commonsense-1","title":"Text-based RL Agents with Commonsense Knowledge: New Challenges, Environments and Approaches","date":"2020-07-12","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"graphopt-learning-optimization-models-of","title":"GraphOpt: Learning Optimization Models of Graph Formation","date":"2020-07-07","arxiv_id":"2007.03619","repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-inverse-reinforcement-learning-using","title":"Learning \"What-if\" Explanations for Sequential Decision-Making","date":"2020-07-02","arxiv_id":"2007.13531","repositories_listed":0,"syntology":null},{"url":null,"slug":"convex-regularization-in-monte-carlo-tree","title":"Convex Regularization in Monte-Carlo Tree Search","date":"2020-07-01","arxiv_id":"2007.00391","repositories_listed":0,"syntology":null},{"url":null,"slug":"falsification-based-robust-adversarial","title":"Falsification-Based Robust Adversarial Reinforcement Learning","date":"2020-07-01","arxiv_id":"2007.00691","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-a-survey","title":"Model-based Reinforcement Learning: A Survey","date":"2020-06-30","arxiv_id":"2006.16712","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-bellman-s-optimality-principle-for-zs","title":"On Bellman's Optimality Principle for zs-POSGs","date":"2020-06-29","arxiv_id":"2006.16395","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-reinforcement-learning-and","title":"A Unifying Framework for Reinforcement Learning and Planning","date":"2020-06-26","arxiv_id":"2006.15009","repositories_listed":0,"syntology":null}],"record_sha256":"a4b067d590f46a528d58f08c7276ee5d0a4441e21e2a5a6065d4559f95e1cef4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}