{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/sequential-decision-making/papers/6","list_of":"/task/sequential-decision-making","task":"Sequential Decision Making","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":13,"rows_per_page":100,"rows":[501,600],"of":1210,"counts":{"archive_papers_tagged":1210,"with_a_code_link":351,"where_syntology_ran_a_sample":107,"not_listed_spam_title":0,"listed":1210,"listed_where_code_ran":107,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":90,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":90,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/sequential-decision-making","prev":"/task/sequential-decision-making/papers/5","next":"/task/sequential-decision-making/papers/7","papers":[{"url":null,"slug":"how-to-measure-human-ai-prediction-accuracy","title":"How to Measure Human-AI Prediction Accuracy in Explainable AI Systems","date":"2024-08-23","arxiv_id":"2409.00069","repositories_listed":0,"syntology":null},{"url":null,"slug":"pareto-inverse-reinforcement-learning-for","title":"Pareto Inverse Reinforcement Learning for Diverse Expert Policy Generation","date":"2024-08-22","arxiv_id":"2408.12110","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-end-to-end-reinforcement-learning-based","title":"An End-to-End Reinforcement Learning Based Approach for Micro-View Order-Dispatching in Ride-Hailing","date":"2024-08-20","arxiv_id":"2408.10479","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-bandits-for-unbounded-context","title":"Contextual Bandits for Unbounded Context Distributions","date":"2024-08-19","arxiv_id":"2408.09655","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-clustering-of-neural-bandits","title":"Meta Clustering of Neural Bandits","date":"2024-08-10","arxiv_id":"2408.05586","repositories_listed":0,"syntology":null},{"url":null,"slug":"structure-and-reduction-of-mcts-for","title":"Structure and Reduction of MCTS for Explainable-AI","date":"2024-08-10","arxiv_id":"2408.05488","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-maximizing-policies-that-fulfill-multi","title":"Non-maximizing policies that fulfill multi-criterion aspirations in expectation","date":"2024-08-08","arxiv_id":"2408.04385","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02949","title":"Few-shot Scooping Under Domain Shift via Simulated Maximal Deployment Gaps","date":"2024-08-06","arxiv_id":"2408.02949","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-choose-a-reinforcement-learning","title":"How to Choose a Reinforcement-Learning Algorithm","date":"2024-07-30","arxiv_id":"2407.20917","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-sustainable-energy","title":"Reinforcement Learning for Sustainable Energy: A Survey","date":"2024-07-26","arxiv_id":"2407.18597","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-ai-rationality-the-random-guesser","title":"Assessing AI Utility: The Random Guesser Test for Sequential Decision-Making Systems","date":"2024-07-25","arxiv_id":"2407.20276","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-quantum-architecture-search-in","title":"Differentiable Quantum Architecture Search in Asynchronous Quantum Reinforcement Learning","date":"2024-07-25","arxiv_id":"2407.18202","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-behavior-cloning-all-you-need","title":"Is Behavior Cloning All You Need? Understanding Horizon in Imitation Learning","date":"2024-07-20","arxiv_id":"2407.15007","repositories_listed":0,"syntology":null},{"url":null,"slug":"managing-risk-using-rolling-forecasts-in","title":"Managing Risk using Rolling Forecasts in Energy-Limited and Stochastic Energy Systems","date":"2024-07-18","arxiv_id":"2407.13626","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-unbound","title":"Exploration Unbound","date":"2024-07-16","arxiv_id":"2407.12178","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretability-in-action-exploratory","title":"Interpretability in Action: Exploratory Analysis of VPT, a Minecraft Agent","date":"2024-07-16","arxiv_id":"2407.12161","repositories_listed":0,"syntology":null},{"url":"/paper/last-iterate-global-convergence-of-policy","slug":"last-iterate-global-convergence-of-policy","title":"Last-Iterate Global Convergence of Policy Gradients for Constrained Reinforcement Learning","date":"2024-07-15","arxiv_id":"2407.10775","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/last-iterate-global-convergence-of-policy#ran","syntology_url":"https://syntology.ai/paper/2407.10775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10775"}},"official":null}},{"url":null,"slug":"predicting-and-understanding-human-action-1","title":"Predicting and Understanding Human Action Decisions: Insights from Large Language Models and Cognitive Instance-Based Learning","date":"2024-07-12","arxiv_id":"2407.09281","repositories_listed":0,"syntology":null},{"url":null,"slug":"mdp-geometry-normalization-and-value-free","title":"MDP Geometry, Normalization and Reward Balancing Solvers","date":"2024-07-09","arxiv_id":"2407.06712","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-and-control-co-design-in-6g","title":"Communication and Control Co-Design in 6G: Sequential Decision-Making with LLMs","date":"2024-07-06","arxiv_id":"2407.06227","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximizing-utility-in-multi-agent","title":"Maximizing utility in multi-agent environments by anticipating the behavior of other learners","date":"2024-07-05","arxiv_id":"2407.04889","repositories_listed":0,"syntology":null},{"url":null,"slug":"short-long-policy-evaluation-with-novel","title":"Short-Long Policy Evaluation with Novel Actions","date":"2024-07-04","arxiv_id":"2407.03674","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-a-physics-informed-decision","title":"Exploring a Physics-Informed Decision Transformer for Distribution System Restoration: Methodology and Performance Analysis","date":"2024-06-30","arxiv_id":"2407.00808","repositories_listed":0,"syntology":null},{"url":null,"slug":"tradeoffs-when-considering-deep-reinforcement","title":"Tradeoffs When Considering Deep Reinforcement Learning for Contingency Management in Advanced Air Mobility","date":"2024-06-28","arxiv_id":"2407.00197","repositories_listed":0,"syntology":null},{"url":null,"slug":"uno-arena-for-evaluating-sequential-decision","title":"UNO Arena for Evaluating Sequential Decision-Making Capability of Large Language Models","date":"2024-06-24","arxiv_id":"2406.16382","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-matrix-diagonalization-through","title":"Accelerating Matrix Diagonalization through Decision Transformers with Epsilon-Greedy Optimization","date":"2024-06-23","arxiv_id":"2406.16191","repositories_listed":0,"syntology":null},{"url":null,"slug":"ardup-active-region-video-diffusion-for","title":"ARDuP: Active Region Video Diffusion for Universal Policies","date":"2024-06-19","arxiv_id":"2406.13301","repositories_listed":0,"syntology":null},{"url":null,"slug":"learned-graph-rewriting-with-equality","title":"Learned Graph Rewriting with Equality Saturation: A New Paradigm in Relational Query Rewrite and Beyond","date":"2024-06-19","arxiv_id":"2407.12794","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-with-2","title":"Constrained Reinforcement Learning with Average Reward Objective: Model-Based and Model-Free Algorithms","date":"2024-06-17","arxiv_id":"2406.11481","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-sequential-decision-making-with","title":"Efficient Sequential Decision Making with Large Language Models","date":"2024-06-17","arxiv_id":"2406.12125","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-adaptation-for-time-constrained-1","title":"Model Adaptation for Time Constrained Embodied Control","date":"2024-06-17","arxiv_id":"2406.11128","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-upper-confidence-bounds-with-near","title":"Data-Driven Upper Confidence Bounds with Near-Optimal Regret for Heavy-Tailed Bandits","date":"2024-06-09","arxiv_id":"2406.05710","repositories_listed":0,"syntology":null},{"url":null,"slug":"rectifying-reinforcement-learning-for-reward","title":"Rectifying Reinforcement Learning for Reward Matching","date":"2024-06-04","arxiv_id":"2406.02213","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-rank-finetuning-for-llms-a-fairness","title":"Low-rank finetuning for LLMs: A fairness perspective","date":"2024-05-28","arxiv_id":"2405.18572","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-offline-data-in-linear-latent","title":"Leveraging Offline Data in Linear Latent Bandits","date":"2024-05-27","arxiv_id":"2405.17324","repositories_listed":0,"syntology":null},{"url":null,"slug":"opera-automatic-offline-policy-evaluation","title":"OPERA: Automatic Offline Policy Evaluation with Re-weighted Aggregates of Multiple Estimators","date":"2024-05-27","arxiv_id":"2405.17708","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-offline-multi-agent-skill","title":"Variational Offline Multi-agent Skill Discovery","date":"2024-05-26","arxiv_id":"2405.16386","repositories_listed":0,"syntology":null},{"url":null,"slug":"inference-of-utilities-and-time-preference-in","title":"Inference of Utilities and Time Preference in Sequential Decision-Making","date":"2024-05-24","arxiv_id":"2405.15975","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-rlignment-inverse-reinforcement","title":"Inverse-RLignment: Large Language Model Alignment from Demonstrations through Inverse Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15624","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-finite-time-analysis-of-distributed-q","title":"A finite time analysis of distributed Q-learning","date":"2024-05-23","arxiv_id":"2405.14078","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcing-language-agents-via-policy","title":"Reinforcing Language Agents via Policy Optimization with Action Decomposition","date":"2024-05-23","arxiv_id":"2405.15821","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-stage-ml-guided-decision-rules-for","title":"Efficiently Training Deep-Learning Parametric Policies using Lagrangian Duality","date":"2024-05-23","arxiv_id":"2405.14973","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-training-and-generalization","title":"Understanding the Training and Generalization of Pretrained Transformer for Sequential Decision Making","date":"2024-05-23","arxiv_id":"2405.14219","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-brittle-foundations-of-react-prompting","title":"On the Brittle Foundations of ReAct Prompting for Agentic Large Language Models","date":"2024-05-22","arxiv_id":"2405.13966","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-linear-programming-framework-for","title":"A Unified Linear Programming Framework for Offline Reward Learning from Human Demonstrations and Feedback","date":"2024-05-20","arxiv_id":"2405.12421","repositories_listed":0,"syntology":null},{"url":null,"slug":"cps-llm-large-language-model-based-safe-usage","title":"CPS-LLM: Large Language Model based Safe Usage Plan Generator for Human-in-the-Loop Human-in-the-Plant Cyber-Physical System","date":"2024-05-19","arxiv_id":"2405.11458","repositories_listed":0,"syntology":null},{"url":null,"slug":"agentclinic-a-multimodal-agent-benchmark-to","title":"AgentClinic: a multimodal agent benchmark to evaluate AI in simulated clinical environments","date":"2024-05-13","arxiv_id":"2405.07960","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-modeling-in-sequential-decision-making","title":"Human-Modeling in Sequential Decision-Making: An Analysis through the Lens of Human-Aware AI","date":"2024-05-13","arxiv_id":"2405.07773","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-q-learning-with-large-language","title":"Enhancing Q-Learning with Large Language Model Heuristics","date":"2024-05-06","arxiv_id":"2405.03341","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-planning-abstractions-from-language","title":"Learning Planning Abstractions from Language","date":"2024-05-06","arxiv_id":"2405.03864","repositories_listed":0,"syntology":null},{"url":null,"slug":"out-of-distribution-adaptation-in-offline-rl","title":"Out-of-Distribution Adaptation in Offline RL: Counterfactual Reasoning via Causal Normalizing Flows","date":"2024-05-06","arxiv_id":"2405.03892","repositories_listed":0,"syntology":null},{"url":null,"slug":"mexgen-an-effective-and-efficient-information","title":"MEXGEN: An Effective and Efficient Information Gain Approximation for Information Gathering Path Planning","date":"2024-05-04","arxiv_id":"2405.02605","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathematics-of-statistical-sequential","title":"Mathematics of statistical sequential decision-making: concentration, risk-awareness and modelling in stochastic bandits, with applications to bariatric surgery","date":"2024-05-03","arxiv_id":"2405.01994","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-reinforcement-learning-for-2","title":"Provably Efficient Reinforcement Learning for Adversarial Restless Multi-Armed Bandits with Unknown Transitions and Bandit Feedback","date":"2024-05-02","arxiv_id":"2405.00950","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-bayesian-inference-in-the-era-of","title":"Scalable Bayesian Inference in the Era of Deep Learning: From Gaussian Processes to Deep Neural Networks","date":"2024-04-29","arxiv_id":"2404.19157","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-to-navigate-turbulence-without-a","title":"Q-learning with temporal memory to navigate turbulence","date":"2024-04-26","arxiv_id":"2404.17495","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-twins-for-forecasting-and-decision","title":"Digital Twins for forecasting and decision optimisation with machine learning: applications in wastewater treatment","date":"2024-04-23","arxiv_id":"2404.14635","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-llms-play-dice-exploring-probability","title":"Do LLMs Play Dice? Exploring Probability Distribution Sampling in Large Language Models for Behavioral Simulation","date":"2024-04-13","arxiv_id":"2404.09043","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-learning-from-suboptimal","title":"Reward Learning from Suboptimal Demonstrations with Applications in Surgical Electrocautery","date":"2024-04-10","arxiv_id":"2404.07185","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularized-conditional-diffusion-model-for","title":"Regularized Conditional Diffusion Model for Multi-Task Preference Alignment","date":"2024-04-07","arxiv_id":"2404.04920","repositories_listed":0,"syntology":null},{"url":null,"slug":"composite-bayesian-optimization-in-function","title":"Composite Bayesian Optimization In Function Spaces Using NEON -- Neural Epistemic Operator Networks","date":"2024-04-03","arxiv_id":"2404.03099","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-granular-adversarial-attacks-against","title":"Multi-granular Adversarial Attacks against Black-box Neural Ranking Models","date":"2024-04-02","arxiv_id":"2404.01574","repositories_listed":0,"syntology":null},{"url":null,"slug":"retentive-decision-transformer-with-adaptive","title":"Retentive Decision Transformer with Adaptive Masking for Reinforcement Learning based Recommendation Systems","date":"2024-03-26","arxiv_id":"2403.17634","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-vision-and-language-navigation","title":"Continual Vision-and-Language Navigation","date":"2024-03-22","arxiv_id":"2403.15049","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-decision-making-for-inline-text","title":"Sequential Decision-Making for Inline Text Autocomplete","date":"2024-03-21","arxiv_id":"2403.15502","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-value-tracking-for-deep-reinforcement","title":"Fast Value Tracking for Deep Reinforcement Learning","date":"2024-03-19","arxiv_id":"2403.13178","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-separated-sarsa-a-practical-sequential","title":"State-Separated SARSA: A Practical Sequential Decision-Making Algorithm with Recovering Rewards","date":"2024-03-18","arxiv_id":"2403.11520","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-fine-tuning-as-inverse","title":"Supervised Fine-Tuning as Inverse Reinforcement Learning","date":"2024-03-18","arxiv_id":"2403.12017","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-multi-objective-dynamic","title":"Distributed Multi-Objective Dynamic Offloading Scheduling for Air-Ground Cooperative MEC","date":"2024-03-16","arxiv_id":"2403.10927","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-minimization-via-saddle-point-1","title":"Regret Minimization via Saddle Point Optimization","date":"2024-03-15","arxiv_id":"2403.10379","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoguide-automated-generation-and-selection","title":"AutoGuide: Automated Generation and Selection of Context-Aware Guidelines for Large Language Model Agents","date":"2024-03-13","arxiv_id":"2403.08978","repositories_listed":0,"syntology":null},{"url":null,"slug":"coral-collaborative-retrieval-augmented-large","title":"CoRAL: Collaborative Retrieval-Augmented Large Language Models Improve Long-tail Recommendation","date":"2024-03-11","arxiv_id":"2403.06447","repositories_listed":0,"syntology":null},{"url":null,"slug":"linearapt-an-adaptive-algorithm-for-the-fixed","title":"LinearAPT: An Adaptive Algorithm for the Fixed-Budget Thresholding Linear Bandit Problem","date":"2024-03-10","arxiv_id":"2403.06230","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-design-of-photonic-crystal-surface","title":"Inverse Design of Photonic Crystal Surface Emitting Lasers is a Sequence Modeling Problem","date":"2024-03-08","arxiv_id":"2403.05149","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-bayesian-optimization-for","title":"Cooperative Bayesian Optimization for Imperfect Agents","date":"2024-03-07","arxiv_id":"2403.04442","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-applications-of-reinforcement","title":"A Survey on Applications of Reinforcement Learning in Spatial Resource Allocation","date":"2024-03-06","arxiv_id":"2403.03643","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-guided-exploration-for-rl-agents-in","title":"Language Guided Exploration for RL Agents in Text Environments","date":"2024-03-05","arxiv_id":"2403.03141","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-learning-rate-for-follow-the","title":"Adaptive Learning Rate for Follow-the-Regularized-Leader: Competitive Analysis and Best-of-Both-Worlds","date":"2024-03-01","arxiv_id":"2403.00715","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-role-of-information-structure-in","title":"On the Role of Information Structure in Reinforcement Learning for Partially-Observable Sequential Teams and Games","date":"2024-03-01","arxiv_id":"2403.00993","repositories_listed":0,"syntology":null},{"url":null,"slug":"successfully-guiding-humans-with-imperfect","title":"Successfully Guiding Humans with Imperfect Instructions by Highlighting Potential Errors and Suggesting Corrections","date":"2024-02-26","arxiv_id":"2402.16973","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-theoretic-safe-bayesian","title":"Information-Theoretic Safe Bayesian Optimization","date":"2024-02-23","arxiv_id":"2402.15347","repositories_listed":0,"syntology":null},{"url":null,"slug":"betail-behavior-transformer-adversarial","title":"BeTAIL: Behavior Transformer Adversarial Imitation Learning from Human Racing Gameplay","date":"2024-02-22","arxiv_id":"2402.14194","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-performance-of-empirical-risk","title":"On the Performance of Empirical Risk Minimization with Smoothed Data","date":"2024-02-22","arxiv_id":"2402.14987","repositories_listed":0,"syntology":null},{"url":null,"slug":"align-your-intents-offline-imitation-learning","title":"Align Your Intents: Offline Imitation Learning via Optimal Transport","date":"2024-02-20","arxiv_id":"2402.13037","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-transformers-revolutionizing-the","title":"Toward TransfORmers: Revolutionizing the Solution of Mixed Integer Programs with Transformers","date":"2024-02-20","arxiv_id":"2402.13380","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-evolving-autoencoder-embedded-q-network","title":"Self-evolving Autoencoder Embedded Q-Network","date":"2024-02-18","arxiv_id":"2402.11604","repositories_listed":0,"syntology":null},{"url":null,"slug":"probability-tools-for-sequential-random","title":"Probability Tools for Sequential Random Projection","date":"2024-02-16","arxiv_id":"2402.14026","repositories_listed":0,"syntology":null},{"url":null,"slug":"auxiliary-reward-generation-with-transition","title":"Auxiliary Reward Generation with Transition Distance Representation Learning","date":"2024-02-12","arxiv_id":"2402.07412","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-sequential-decision-making-with","title":"Online Sequential Decision-Making with Unknown Delays","date":"2024-02-12","arxiv_id":"2402.07703","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-risk-sensitive-rl-with-partial","title":"Offline Risk-sensitive RL with Partial Observability to Enhance Performance in Human-Robot Teaming","date":"2024-02-08","arxiv_id":"2402.05703","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for-dynamic","title":"A Reinforcement Learning Approach for Dynamic Rebalancing in Bike-Sharing System","date":"2024-02-05","arxiv_id":"2402.03589","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-21","title":"Multi-Agent Reinforcement Learning for Offloading Cellular Communications with Cooperating UAVs","date":"2024-02-05","arxiv_id":"2402.02957","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularized-q-learning-with-linear-function","title":"Regularized Q-Learning with Linear Function Approximation","date":"2024-01-26","arxiv_id":"2401.15196","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-dynamic-power-dispatch-with-high","title":"Stochastic Dynamic Power Dispatch with High Generalization and Few-Shot Adaption via Contextual Meta Graph Reinforcement Learning","date":"2024-01-19","arxiv_id":"2401.12235","repositories_listed":0,"syntology":null},{"url":null,"slug":"llms-for-relational-reasoning-how-far-are-we","title":"LLMs for Relational Reasoning: How Far are We?","date":"2024-01-17","arxiv_id":"2401.09042","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-off-policy-reinforcement-learning-for","title":"Towards Off-Policy Reinforcement Learning for Ranking Policies with Human Feedback","date":"2024-01-17","arxiv_id":"2401.08959","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-q-learning-for-combinatorial","title":"Graph Q-Learning for Combinatorial Optimization","date":"2024-01-11","arxiv_id":"2401.05610","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactions-between-dynamic-team-composition","title":"Interactions between dynamic team composition and coordination: An agent-based modeling approach","date":"2024-01-11","arxiv_id":"2401.05832","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-sample-efficient-offline-reinforcement-1","title":"On Sample-Efficient Offline Reinforcement Learning: Data Diversity, Posterior Sampling, and Beyond","date":"2024-01-06","arxiv_id":"2401.03301","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-an-adaptable-and-generalizable","title":"Towards an Adaptable and Generalizable Optimization Engine in Decision and Control: A Meta Reinforcement Learning Approach","date":"2024-01-04","arxiv_id":"2401.02508","repositories_listed":0,"syntology":null}],"record_sha256":"01d50e13c5aa56dae734d2f6e44744c5d1cb1ebe33501f73b37371e9625d458b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}