{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/sequential-decision-making/papers/5","list_of":"/task/sequential-decision-making","task":"Sequential Decision Making","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":13,"rows_per_page":100,"rows":[401,500],"of":1210,"counts":{"archive_papers_tagged":1210,"with_a_code_link":351,"where_syntology_ran_a_sample":107,"not_listed_spam_title":0,"listed":1210,"listed_where_code_ran":107,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":90,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":90,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/sequential-decision-making","prev":"/task/sequential-decision-making/papers/4","next":"/task/sequential-decision-making/papers/6","papers":[{"url":null,"slug":"relaxing-the-markov-requirements-on","title":"A Framework of decision-relevant observability: Reinforcement Learning converges under relative ignorability","date":"2025-04-10","arxiv_id":"2504.07722","repositories_listed":0,"syntology":null},{"url":null,"slug":"raise-reinforenced-adaptive-instruction","title":"RAISE: Reinforenced Adaptive Instruction Selection For Large Language Models","date":"2025-04-09","arxiv_id":"2504.07282","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-classification-view-on-meta-learning","title":"A Classification View on Meta Learning Bandits","date":"2025-04-06","arxiv_id":"2504.04505","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-automation-to-autonomy-in-smart","title":"From Automation to Autonomy in Smart Manufacturing: A Bayesian Optimization Framework for Modeling Multi-Objective Experimentation and Sequential Decision Making","date":"2025-04-05","arxiv_id":"2504.04244","repositories_listed":0,"syntology":null},{"url":null,"slug":"moral-a-multimodal-reinforcement-learning","title":"MORAL: A Multimodal Reinforcement Learning Framework for Decision Making in Autonomous Laboratories","date":"2025-04-04","arxiv_id":"2504.03153","repositories_listed":0,"syntology":null},{"url":null,"slug":"counterfactual-inference-under-thompson","title":"Counterfactual Inference under Thompson Sampling","date":"2025-04-03","arxiv_id":"2504.08773","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-enabling-learning-for-time-varying","title":"Towards Enabling Learning for Time-Varying finite horizon Sequential Decision-Making Problems*","date":"2025-04-02","arxiv_id":"2504.02129","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-evaluation-for-sequential","title":"Off-Policy Evaluation for Sequential Persuasion Process with Unobserved Confounding","date":"2025-04-01","arxiv_id":"2504.01211","repositories_listed":0,"syntology":null},{"url":null,"slug":"remember-but-also-forget-bridging-myopic-and","title":"Remember, but also, Forget: Bridging Myopic and Perfect Recall Fairness with Past-Discounting","date":"2025-04-01","arxiv_id":"2504.01154","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-explainable-multi-player-mcts","title":"Exploring Explainable Multi-player MCTS-minimax Hybrids in Board Game Using Process Mining","date":"2025-03-30","arxiv_id":"2503.23326","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-action-free-learning-of-ex-bmdps-by","title":"Offline Action-Free Learning of Ex-BMDPs by Comparing Diverse Datasets","date":"2025-03-26","arxiv_id":"2503.21018","repositories_listed":0,"syntology":null},{"url":null,"slug":"observation-adaptation-via-annealed","title":"Observation Adaptation via Annealed Importance Resampling for Partially Observable Markov Decision Processes","date":"2025-03-25","arxiv_id":"2503.19302","repositories_listed":0,"syntology":null},{"url":null,"slug":"viper-visual-perception-and-explainable","title":"VIPER: Visual Perception and Explainable Reasoning for Sequential Decision-Making","date":"2025-03-19","arxiv_id":"2503.15108","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-in-visual-navigation-of-end-to-end","title":"Reasoning in visual navigation of end-to-end trained agents: a dynamical systems approach","date":"2025-03-11","arxiv_id":"2503.08306","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-action-generalization-with-limited","title":"Zero-Shot Action Generalization with Limited Observations","date":"2025-03-11","arxiv_id":"2503.08867","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-dependent-regret-bounds-in-multi-armed","title":"Graph-Dependent Regret Bounds in Multi-Armed Bandits with Interference","date":"2025-03-10","arxiv_id":"2503.07555","repositories_listed":0,"syntology":null},{"url":null,"slug":"gflowvlm-enhancing-multi-step-reasoning-in","title":"GFlowVLM: Enhancing Multi-step Reasoning in Vision-Language Models with Generative Flow Networks","date":"2025-03-09","arxiv_id":"2503.06514","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-graph-traversal","title":"Bayesian Graph Traversal","date":"2025-03-07","arxiv_id":"2503.05963","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-01734","title":"Adversarial Agents: Black-Box Evasion Attacks with Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.01734","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-parametric-batched-global-multi-armed","title":"Semi-Parametric Batched Global Multi-Armed Bandits with Covariates","date":"2025-03-01","arxiv_id":"2503.00565","repositories_listed":0,"syntology":null},{"url":null,"slug":"shaping-laser-pulses-with-reinforcement","title":"Shaping Laser Pulses with Reinforcement Learning","date":"2025-03-01","arxiv_id":"2503.00499","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-evolving-landscape-of-llm-and-vlm","title":"The Evolving Landscape of LLM- and VLM-Integrated Reinforcement Learning","date":"2025-02-21","arxiv_id":"2502.15214","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-ultrasound-image","title":"Reinforcement Learning for Ultrasound Image Analysis A Comprehensive Review of Advances and Applications","date":"2025-02-20","arxiv_id":"2502.14995","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-fair-policies-for-infectious","title":"Learning Fair Policies for Infectious Diseases Mitigation using Path Integral Control","date":"2025-02-14","arxiv_id":"2502.09831","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-evaluation-for-job-shop-scheduling","title":"Self-Evaluation for Job-Shop Scheduling","date":"2025-02-12","arxiv_id":"2502.08684","repositories_listed":0,"syntology":null},{"url":null,"slug":"vsc-rl-advancing-autonomous-vision-language","title":"Advancing Autonomous VLM Agents via Variational Subgoal-Conditioned Reinforcement Learning","date":"2025-02-11","arxiv_id":"2502.07949","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-explainable-deep-reinforcement","title":"A Survey on Explainable Deep Reinforcement Learning","date":"2025-02-08","arxiv_id":"2502.06869","repositories_listed":0,"syntology":null},{"url":null,"slug":"unifying-and-optimizing-data-values-for","title":"Unifying and Optimizing Data Values for Selection via Sequential-Decision-Making","date":"2025-02-06","arxiv_id":"2502.04554","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-clustering-of-dueling-bandits","title":"Online Clustering of Dueling Bandits","date":"2025-02-04","arxiv_id":"2502.02079","repositories_listed":0,"syntology":null},{"url":null,"slug":"vla-cache-towards-efficient-vision-language","title":"VLA-Cache: Towards Efficient Vision-Language-Action Model via Adaptive Token Caching in Robotic Manipulation","date":"2025-02-04","arxiv_id":"2502.02175","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-prompt-optimization-for-llm-based","title":"Meta-Prompt Optimization for LLM-Based Sequential Decision Making","date":"2025-02-02","arxiv_id":"2502.00728","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-learning-for-combinatorial-multi","title":"Offline Learning for Combinatorial Multi-armed Bandits","date":"2025-01-31","arxiv_id":"2501.19300","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-online-decision-making-with","title":"Contextual Online Decision Making with Infinite-Dimensional Functional Regression","date":"2025-01-30","arxiv_id":"2501.18359","repositories_listed":0,"syntology":null},{"url":null,"slug":"deceptive-sequential-decision-making-via","title":"Deceptive Sequential Decision-Making via Regularized Policy Optimization","date":"2025-01-30","arxiv_id":"2501.18803","repositories_listed":0,"syntology":null},{"url":null,"slug":"hd-cb-the-first-exploration-of","title":"HD-CB: The First Exploration of Hyperdimensional Computing for Contextual Bandits Problems","date":"2025-01-28","arxiv_id":"2501.16863","repositories_listed":0,"syntology":null},{"url":"/paper/sample-efficient-behavior-cloning-using","slug":"sample-efficient-behavior-cloning-using","title":"Sample-Efficient Behavior Cloning Using General Domain Knowledge","date":"2025-01-27","arxiv_id":"2501.16546","repositories_listed":0,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/sample-efficient-behavior-cloning-using#ran","syntology_url":"https://syntology.ai/paper/2501.16546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.16546"}},"official":null}},{"url":null,"slug":"bridging-visualization-and-optimization","title":"Bridging Visualization and Optimization: Multimodal Large Language Models on Graph-Structured Combinatorial Optimization","date":"2025-01-21","arxiv_id":"2501.11968","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-scene-understanding-for-vision","title":"Embodied Scene Understanding for Vision Language Models via MetaVQA","date":"2025-01-15","arxiv_id":"2501.09167","repositories_listed":0,"syntology":null},{"url":null,"slug":"all-ai-models-are-wrong-but-some-are-optimal","title":"All AI Models are Wrong, but Some are Optimal","date":"2025-01-10","arxiv_id":"2501.06086","repositories_listed":0,"syntology":null},{"url":null,"slug":"counterfactually-fair-reinforcement-learning","title":"Counterfactually Fair Reinforcement Learning via Sequential Data Preprocessing","date":"2025-01-10","arxiv_id":"2501.06366","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-flow-networks-theory-and","title":"Generative Flow Networks: Theory and Applications to Structure Learning","date":"2025-01-09","arxiv_id":"2501.05498","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-reinforcement-learning-via","title":"Explainable Reinforcement Learning via Temporal Policy Decomposition","date":"2025-01-07","arxiv_id":"2501.03902","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-mathcal-o-sqrt-t-regret-decoupling","title":"Beyond $\\mathcal{O}(\\sqrt{T})$ Regret: Decoupling Learning and Decision-making in Online Linear Programming","date":"2025-01-06","arxiv_id":"2501.02761","repositories_listed":0,"syntology":null},{"url":null,"slug":"demystifying-online-clustering-of-bandits","title":"Demystifying Online Clustering of Bandits: Enhanced Exploration Under Stochastic and Smoothed Adversarial Contexts","date":"2025-01-01","arxiv_id":"2501.00891","repositories_listed":0,"syntology":null},{"url":null,"slug":"seqmvrl-a-sequential-fusion-framework-for","title":"SeqMvRL: A Sequential Fusion Framework for Multi-view Representation Learning","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-fantasy-sports-team-selection-with","title":"Optimizing Fantasy Sports Team Selection with Deep Reinforcement Learning","date":"2024-12-26","arxiv_id":"2412.19215","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperq-opt-q-learning-for-hyperparameter","title":"HyperQ-Opt: Q-learning for Hyperparameter Optimization","date":"2024-12-23","arxiv_id":"2412.17765","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairness-in-reinforcement-learning-with","title":"Fairness in Reinforcement Learning with Bisimulation Metrics","date":"2024-12-22","arxiv_id":"2412.17123","repositories_listed":0,"syntology":null},{"url":null,"slug":"gas-generative-auto-bidding-with-post","title":"GAS: Generative Auto-bidding with Post-training Search","date":"2024-12-22","arxiv_id":"2412.17018","repositories_listed":0,"syntology":null},{"url":null,"slug":"subgoal-discovery-using-a-free-energy","title":"Subgoal Discovery Using a Free Energy Paradigm and State Aggregations","date":"2024-12-21","arxiv_id":"2412.16687","repositories_listed":0,"syntology":null},{"url":null,"slug":"drivegpt-scaling-autoregressive-behavior","title":"DriveGPT: Scaling Autoregressive Behavior Models for Driving","date":"2024-12-19","arxiv_id":"2412.14415","repositories_listed":0,"syntology":null},{"url":null,"slug":"threshold-uct-cost-constrained-monte-carlo","title":"Threshold UCT: Cost-Constrained Monte Carlo Tree Search with Pareto Curves","date":"2024-12-18","arxiv_id":"2412.13962","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-reinforcement-learning-strategies-for","title":"Active Reinforcement Learning Strategies for Offline Policy Improvement","date":"2024-12-17","arxiv_id":"2412.13106","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-robust-markov-decision-processes","title":"Solving Robust Markov Decision Processes: Generic, Reliable, Efficient","date":"2024-12-13","arxiv_id":"2412.10185","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-reward-specification-in-deep","title":"Effective Reward Specification in Deep Reinforcement Learning","date":"2024-12-10","arxiv_id":"2412.07177","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-sensor-redundancy-in-sequential","title":"Optimizing Sensor Redundancy in Sequential Decision-Making Problems","date":"2024-12-10","arxiv_id":"2412.07686","repositories_listed":0,"syntology":null},{"url":null,"slug":"swarm-behavior-cloning","title":"Swarm Behavior Cloning","date":"2024-12-10","arxiv_id":"2412.07617","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-note-on-sample-complexity-of-interactive","title":"A Note on Sample Complexity of Interactive Imitation Learning with Log Loss","date":"2024-12-09","arxiv_id":"2412.07057","repositories_listed":0,"syntology":null},{"url":null,"slug":"conservative-contextual-bandits-beyond-linear","title":"Conservative Contextual Bandits: Beyond Linear Representations","date":"2024-12-09","arxiv_id":"2412.06165","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrated-sensing-and-communications-for-low","title":"Integrated Sensing and Communications for Low-Altitude Economy: A Deep Reinforcement Learning Approach","date":"2024-12-05","arxiv_id":"2412.04074","repositories_listed":0,"syntology":null},{"url":null,"slug":"failure-probability-estimation-for-black-box","title":"Failure Probability Estimation for Black-Box Autonomous Systems using State-Dependent Importance Sampling Proposals","date":"2024-12-03","arxiv_id":"2412.02154","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-reviews-of-bandit-problems-in-ai","title":"Selective Reviews of Bandit Problems in AI via a Statistical View","date":"2024-12-03","arxiv_id":"2412.02251","repositories_listed":0,"syntology":null},{"url":null,"slug":"technical-report-on-reinforcement-learning","title":"Technical Report on Reinforcement Learning Control on the Lucas-Nülle Inverted Pendulum","date":"2024-12-03","arxiv_id":"2412.02264","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-series-informed-closed-loop-learning-for","title":"Time-Series-Informed Closed-loop Learning for Sequential Decision Making and Control","date":"2024-12-03","arxiv_id":"2412.02423","repositories_listed":0,"syntology":null},{"url":null,"slug":"steve-audio-expanding-the-goal-conditioning","title":"STEVE-Audio: Expanding the Goal Conditioning Modalities of Embodied Agents in Minecraft","date":"2024-12-01","arxiv_id":"2412.00949","repositories_listed":0,"syntology":null},{"url":null,"slug":"market-making-without-regret","title":"Market Making without Regret","date":"2024-11-21","arxiv_id":"2411.13993","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-markov-decision-processes-a-place","title":"Robust Markov Decision Processes: A Place Where AI and Formal Methods Meet","date":"2024-11-18","arxiv_id":"2411.11451","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-sample-efficiency-and-generalization","title":"Towards Sample-Efficiency and Generalization of Transfer and Inverse Reinforcement Learning: A Comprehensive Literature Review","date":"2024-11-15","arxiv_id":"2411.10268","repositories_listed":0,"syntology":null},{"url":null,"slug":"fair-resource-allocation-in-weakly-coupled","title":"Fair Resource Allocation in Weakly Coupled Markov Decision Processes","date":"2024-11-14","arxiv_id":"2411.09804","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-and-federated-black-box","title":"Collaborative and Federated Black-box Optimization: A Bayesian Optimization Perspective","date":"2024-11-12","arxiv_id":"2411.07523","repositories_listed":0,"syntology":null},{"url":null,"slug":"earl-bo-reinforcement-learning-for-multi-step","title":"EARL-BO: Reinforcement Learning for Multi-Step Lookahead, High-Dimensional Bayesian Optimization","date":"2024-10-31","arxiv_id":"2411.00171","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-reinforcement-learning-based-two","title":"Quantum Reinforcement Learning-Based Two-Stage Unit Commitment Framework for Enhanced Power Systems Robustness","date":"2024-10-28","arxiv_id":"2410.21240","repositories_listed":0,"syntology":null},{"url":null,"slug":"annotation-efficiency-identifying-hard","title":"Annotation Efficiency: Identifying Hard Samples via Blocked Sparse Linear Bandits","date":"2024-10-26","arxiv_id":"2410.20041","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-aligning-large","title":"Reinforcement Learning for Aligning Large Language Models Agents with Interactive Environments: Quantifying and Mitigating Prompt Overfitting","date":"2024-10-25","arxiv_id":"2410.19920","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-thompson-sampling-algorithms-against","title":"Robust Thompson Sampling Algorithms Against Reward Poisoning Attacks","date":"2024-10-25","arxiv_id":"2410.19705","repositories_listed":0,"syntology":null},{"url":null,"slug":"convex-markov-games-a-framework-for-fairness","title":"Convex Markov Games: A New Frontier for Multi-Agent Reinforcement Learning","date":"2024-10-22","arxiv_id":"2410.16600","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-upper-confidence-bounds-for","title":"Hierarchical Upper Confidence Bounds for Constrained Online Learning","date":"2024-10-22","arxiv_id":"2410.17216","repositories_listed":0,"syntology":null},{"url":null,"slug":"sac-glam-improving-online-rl-for-llm-agents","title":"SAC-GLAM: Improving Online RL for LLM agents with Soft Actor-Critic and Hindsight Relabeling","date":"2024-10-16","arxiv_id":"2410.12481","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-control-codesign-for-large","title":"Communication-Control Codesign for Large-Scale Wireless Networked Control Systems","date":"2024-10-15","arxiv_id":"2410.11316","repositories_listed":0,"syntology":null},{"url":null,"slug":"burning-red-unlocking-subtask-driven","title":"Burning RED: Unlocking Subtask-Driven Reinforcement Learning and Risk-Awareness in Average-Reward Markov Decision Processes","date":"2024-10-14","arxiv_id":"2410.10578","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-with-large","title":"Efficient Reinforcement Learning with Large Language Model Priors","date":"2024-10-10","arxiv_id":"2410.07927","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-hierarchical-reinforcement-learning","title":"Offline Hierarchical Reinforcement Learning via Inverse Optimization","date":"2024-10-10","arxiv_id":"2410.07933","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-modeling-capabilities-of-large","title":"On the Modeling Capabilities of Large Language Models for Sequential Decision Making","date":"2024-10-08","arxiv_id":"2410.05656","repositories_listed":0,"syntology":null},{"url":"/paper/dopl-direct-online-preference-learning-for","slug":"dopl-direct-online-preference-learning-for","title":"DOPL: Direct Online Preference Learning for Restless Bandits with Preference Feedback","date":"2024-10-07","arxiv_id":"2410.05527","repositories_listed":0,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dopl-direct-online-preference-learning-for#ran","syntology_url":"https://syntology.ai/paper/2410.05527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05527"}},"official":null}},{"url":null,"slug":"preference-optimization-as-probabilistic","title":"Preference Optimization as Probabilistic Inference","date":"2024-10-05","arxiv_id":"2410.04166","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-optimal-trust-aware-multi-armed","title":"Minimax-optimal trust-aware multi-armed bandits","date":"2024-10-04","arxiv_id":"2410.03651","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-time-varying-optimization-based-on","title":"Safe Time-Varying Optimization based on Gaussian Processes with Spatio-Temporal Kernel","date":"2024-09-26","arxiv_id":"2409.18000","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-utilities-from-demonstrations-in","title":"Learning Utilities from Demonstrations in Markov Decision Processes","date":"2024-09-25","arxiv_id":"2409.17355","repositories_listed":0,"syntology":null},{"url":null,"slug":"pressure-reference-points-and-risk-taking","title":"Reference Points, Risk-Taking Behavior, and Competitive Outcomes in Sequential Settings","date":"2024-09-20","arxiv_id":"2409.13333","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-language-tracking-with-multi-modal","title":"Visual Language Tracking with Multi-modal Interaction: A Robust Benchmark","date":"2024-09-13","arxiv_id":"2409.08887","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-10","title":"Hierarchical Reinforcement Learning for Temporal Abstraction of Listwise Recommendation","date":"2024-09-11","arxiv_id":"2409.07416","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierllm-hierarchical-large-language-model-for","title":"HierLLM: Hierarchical Large Language Model for Question Recommendation","date":"2024-09-10","arxiv_id":"2409.06177","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-introduction-to-quantum-reinforcement","title":"An Introduction to Quantum Reinforcement Learning (QRL)","date":"2024-09-09","arxiv_id":"2409.05846","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-rested-and-restless-bandits-with","title":"Bridging Rested and Restless Bandits with Graph-Triggering: Rising and Rotting","date":"2024-09-09","arxiv_id":"2409.05980","repositories_listed":0,"syntology":null},{"url":null,"slug":"forward-kl-regularized-preference","title":"Forward KL Regularized Preference Optimization for Aligning Diffusion Policies","date":"2024-09-09","arxiv_id":"2409.05622","repositories_listed":0,"syntology":null},{"url":null,"slug":"sliding-window-thompson-sampling-for-non","title":"Sliding-Window Thompson Sampling for Non-Stationary Settings","date":"2024-09-08","arxiv_id":"2409.05181","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-naive-aggregation-algorithm-for-improving","title":"A naive aggregation algorithm for improving generalization in a class of learning problems","date":"2024-09-06","arxiv_id":"2409.04352","repositories_listed":0,"syntology":null},{"url":null,"slug":"infralib-enabling-reinforcement-learning-and","title":"InfraLib: Enabling Reinforcement Learning and Decision-Making for Large-Scale Infrastructure Management","date":"2024-09-05","arxiv_id":"2409.03167","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-sequential-decision-making-model-for","title":"A Sequential Decision-Making Model for Perimeter Identification","date":"2024-09-04","arxiv_id":"2409.02549","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-elections-welfare-strategyproofness","title":"Temporal Elections: Welfare, Strategyproofness, and Proportionality","date":"2024-08-24","arxiv_id":"2408.13637","repositories_listed":0,"syntology":null}],"record_sha256":"74bbfa73e6c62b97cc2b78e6d0cb1c9495e05f0d412ad111207b7be6d866b62b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}