{"url":"/task/sequential-decision-making","name":"Sequential Decision Making","slug":"sequential-decision-making","description_markdown":null,"categories":[{"name":"Methodology","url":"/area/methodology"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":1210,"papers_with_code":351,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/benyfits","name":"BeNYfits","full_name":"New York City Public Benefits Eligibility Dialog Agent Benchmark","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":351,"tagged_in_all":1210,"items":[{"url":"/paper/deep-reinforcement-learning-for-unsupervised","title":"Deep Reinforcement Learning for Unsupervised Video Summarization with Diversity-Representativeness Reward","date":"2017-12-29","arxiv_id":"1801.00054","repositories_listed":6,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/reflexion-language-agents-with-verbal","title":"Reflexion: Language Agents with Verbal Reinforcement Learning","date":"2023-03-20","arxiv_id":"2303.11366","repositories_listed":5,"syntology":{"n":9,"n_ran":2,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/iq-learn-inverse-soft-q-learning-for","title":"IQ-Learn: Inverse soft-Q Learning for Imitation","date":"2021-06-23","arxiv_id":"2106.12142","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/deep-reinforcement-learning-based","title":"Deep Reinforcement Learning based Recommendation with Explicit User-Item Interactions Modeling","date":"2018-10-29","arxiv_id":"1810.12027","repositories_listed":5,"syntology":null},{"url":"/paper/deep-bayesian-bandits-showdown-an-empirical","title":"Deep Bayesian Bandits Showdown: An Empirical Comparison of Bayesian Deep Networks for Thompson Sampling","date":"2018-02-26","arxiv_id":"1802.09127","repositories_listed":4,"syntology":null},{"url":"/paper/learning-multi-level-hierarchies-with","title":"Learning Multi-Level Hierarchies with Hindsight","date":"2017-12-04","arxiv_id":"1712.00948","repositories_listed":4,"syntology":null},{"url":"/paper/thinking-fast-and-slow-with-deep-learning-and","title":"Thinking Fast and Slow with Deep Learning and Tree Search","date":"2017-05-23","arxiv_id":"1705.08439","repositories_listed":4,"syntology":null},{"url":"/paper/is-reinforcement-learning-not-for-natural","title":"Is Reinforcement Learning (Not) for Natural Language Processing: Benchmarks, Baselines, and Building Blocks for Natural Language Policy Optimization","date":"2022-10-03","arxiv_id":"2210.01241","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-imbalanced","title":"Deep Reinforcement Learning for Imbalanced Classification","date":"2019-01-05","arxiv_id":"1901.01379","repositories_listed":3,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/a2-rl-aesthetics-aware-reinforcement-learning","title":"A2-RL: Aesthetics Aware Reinforcement Learning for Image Cropping","date":"2017-09-14","arxiv_id":"1709.04595","repositories_listed":3,"syntology":null},{"url":"/paper/an-alternative-softmax-operator-for","title":"An Alternative Softmax Operator for Reinforcement Learning","date":"2016-12-16","arxiv_id":"1612.05628","repositories_listed":3,"syntology":null},{"url":"/paper/model-free-episodic-control","title":"Model-Free Episodic Control","date":"2016-06-14","arxiv_id":"1606.04460","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/multi-agent-reinforcement-learning-for-22","title":"Multi-Agent Reinforcement Learning for Autonomous Driving: A Survey","date":"2024-08-19","arxiv_id":"2408.09675","repositories_listed":2,"syntology":null},{"url":"/paper/adaptive-foundation-models-for-online","title":"Scalable Exploration via Ensemble++","date":"2024-07-18","arxiv_id":"2407.13195","repositories_listed":2,"syntology":{"n":8,"n_ran":5,"n_unverified":3,"n_pointer_only":8}},{"url":"/paper/sym-q-adaptive-symbolic-regression-via","title":"Sym-Q: Adaptive Symbolic Regression via Sequential Decision-Making","date":"2024-02-07","arxiv_id":"2402.05306","repositories_listed":2,"syntology":null},{"url":"/paper/zero-shot-reinforcement-learning-via-function","title":"Zero-Shot Reinforcement Learning via Function Encoders","date":"2024-01-30","arxiv_id":"2401.17173","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/equal-long-term-benefit-rate-adapting-static","title":"Adapting Static Fairness to Sequential Decision-Making: Bias Mitigation Strategies towards Equal Long-term Benefit Rate","date":"2023-09-07","arxiv_id":"2309.03426","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/learning-embeddings-for-sequential-tasks","title":"Learning Embeddings for Sequential Tasks Using Population of Agents","date":"2023-06-05","arxiv_id":"2306.03311","repositories_listed":2,"syntology":null},{"url":"/paper/trieste-efficiently-exploring-the-depths-of","title":"Trieste: Efficiently Exploring The Depths of Black-box Functions with TensorFlow","date":"2023-02-16","arxiv_id":"2302.08436","repositories_listed":2,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/temporal-latent-bottleneck-synthesis-of-fast","title":"Temporal Latent Bottleneck: Synthesis of Fast and Slow Processing Mechanisms in Sequence Learning","date":"2022-05-30","arxiv_id":"2205.14794","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/efficient-symptom-inquiring-and-diagnosis-via","title":"Efficient Symptom Inquiring and Diagnosis via Adaptive Alignment of Reinforcement Learning and Classification","date":"2021-12-01","arxiv_id":"2112.00733","repositories_listed":2,"syntology":null},{"url":"/paper/mixed-policy-gradient","title":"Mixed Policy Gradient: off-policy reinforcement learning driven jointly by data and model","date":"2021-02-23","arxiv_id":"2102.11513","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/text-based-rl-agents-with-commonsense","title":"Text-based RL Agents with Commonsense Knowledge: New Challenges, Environments and Baselines","date":"2020-10-08","arxiv_id":"2010.03790","repositories_listed":2,"syntology":{"n":13,"n_ran":0,"n_unverified":13,"n_pointer_only":0}},{"url":"/paper/pddlgym-gym-environments-from-pddl-problems","title":"PDDLGym: Gym Environments from PDDL Problems","date":"2020-02-15","arxiv_id":"2002.06432","repositories_listed":2,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/policy-learning-for-malaria-control","title":"Policy Learning for Malaria Control","date":"2019-10-20","arxiv_id":"1910.08926","repositories_listed":2,"syntology":null},{"url":"/paper/classification-with-costly-features-as-a","title":"Classification with Costly Features as a Sequential Decision-Making Problem","date":"2019-09-05","arxiv_id":"1909.02564","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/skipnet-learning-dynamic-routing-in","title":"SkipNet: Learning Dynamic Routing in Convolutional Networks","date":"2017-11-26","arxiv_id":"1711.09485","repositories_listed":2,"syntology":null},{"url":"/paper/detecting-adversarial-attacks-on-neural","title":"Detecting Adversarial Attacks on Neural Network Policies with Visual Foresight","date":"2017-10-02","arxiv_id":"1710.00814","repositories_listed":2,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":10}},{"url":"/paper/learning-model-based-planning-from-scratch","title":"Learning model-based planning from scratch","date":"2017-07-19","arxiv_id":"1707.06170","repositories_listed":2,"syntology":null},{"url":"/paper/doubly-robust-off-policy-value-evaluation-for","title":"Doubly Robust Off-policy Value Evaluation for Reinforcement Learning","date":"2015-11-11","arxiv_id":"1511.03722","repositories_listed":2,"syntology":null}],"syntology_records":15,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}