{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/48","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":48,"pages_in_order":135,"rows_per_page":100,"rows":[4701,4800],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/47","next":"/task/reinforcement-learning-2/papers/49","papers":[{"url":null,"slug":"reservoir-computing-for-fast-simplified","title":"Reservoir Computing for Fast, Simplified Reinforcement Learning on Memory Tasks","date":"2024-12-17","arxiv_id":"2412.13093","repositories_listed":0,"syntology":null},{"url":null,"slug":"equivariant-action-sampling-for-reinforcement","title":"Equivariant Action Sampling for Reinforcement Learning and Planning","date":"2024-12-16","arxiv_id":"2412.12237","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-bayesian-deep-reinforcement","title":"Generalized Bayesian deep reinforcement learning","date":"2024-12-16","arxiv_id":"2412.11743","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-meta-reinforcement-learning-via","title":"Hierarchical Meta-Reinforcement Learning via Automated Macro-Action Discovery","date":"2024-12-16","arxiv_id":"2412.11930","repositories_listed":0,"syntology":null},{"url":null,"slug":"stabilizing-reinforcement-learning-in","title":"Stabilizing Reinforcement Learning in Differentiable Multiphysics Simulation","date":"2024-12-16","arxiv_id":"2412.12089","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-using-finite","title":"Safe Reinforcement Learning using Finite-Horizon Gradient-based Estimation","date":"2024-12-15","arxiv_id":"2412.11138","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-driving-with-evolution-capability-a","title":"Automated Driving with Evolution Capability: A Reinforcement Learning Method with Monotonic Performance Enhancement","date":"2024-12-14","arxiv_id":"2412.10822","repositories_listed":0,"syntology":null},{"url":null,"slug":"lyapunov-based-reinforcement-learning-for-1","title":"Lyapunov-based reinforcement learning for distributed control with stability guarantee","date":"2024-12-14","arxiv_id":"2412.10844","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-scalable","title":"Deep Reinforcement Learning for Scalable Multiagent Spacecraft Inspection","date":"2024-12-13","arxiv_id":"2412.10530","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-instrument-design-with-reinforcement","title":"Physics Instrument Design with Reinforcement Learning","date":"2024-12-13","arxiv_id":"2412.10237","repositories_listed":0,"syntology":null},{"url":null,"slug":"rldg-robotic-generalist-policy-distillation","title":"RLDG: Robotic Generalist Policy Distillation via Reinforcement Learning","date":"2024-12-13","arxiv_id":"2412.09858","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-for-optimal","title":"Efficient Reinforcement Learning for Optimal Control with Natural Images","date":"2024-12-12","arxiv_id":"2412.08893","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-train-based-distributed-multi-agent","title":"Quantum-Train-Based Distributed Multi-Agent Reinforcement Learning","date":"2024-12-12","arxiv_id":"2412.08845","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-sketch-decompositions-in-planning","title":"Learning Sketch Decompositions in Planning via Deep Reinforcement Learning","date":"2024-12-11","arxiv_id":"2412.08574","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-reward-specification-in-deep","title":"Effective Reward Specification in Deep Reinforcement Learning","date":"2024-12-10","arxiv_id":"2412.07177","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-progressive-influence-maximization-in","title":"Non-Progressive Influence Maximization in Dynamic Social Networks","date":"2024-12-10","arxiv_id":"2412.07402","repositories_listed":0,"syntology":null},{"url":null,"slug":"parseval-regularization-for-continual","title":"Parseval Regularization for Continual Reinforcement Learning","date":"2024-12-10","arxiv_id":"2412.07224","repositories_listed":0,"syntology":null},{"url":null,"slug":"edge-delayed-deep-deterministic-policy","title":"Edge Delayed Deep Deterministic Policy Gradient: efficient continuous control for edge scenarios","date":"2024-12-09","arxiv_id":"2412.06390","repositories_listed":0,"syntology":null},{"url":null,"slug":"skill-enhanced-reinforcement-learning","title":"Skill-Enhanced Reinforcement Learning Acceleration from Demonstrations","date":"2024-12-09","arxiv_id":"2412.06207","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-based-deep-reinforcement-learning-of","title":"Vision-Based Deep Reinforcement Learning of UAV Autonomous Navigation Using Privileged Information","date":"2024-12-09","arxiv_id":"2412.06313","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-shaped-prediction-avoiding","title":"Policy-shaped prediction: avoiding distractions in model-based reinforcement learning","date":"2024-12-08","arxiv_id":"2412.05766","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategizing-equitable-transit-evacuations-a","title":"Strategizing Equitable Transit Evacuations: A Data-Driven Reinforcement Learning Approach","date":"2024-12-08","arxiv_id":"2412.05777","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-soft-driving-constraints-from","title":"Learning Soft Driving Constraints from Vectorized Scene Embeddings while Imitating Expert Trajectories","date":"2024-12-07","arxiv_id":"2412.05717","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-temporally-correlated-latent-exploration","title":"A Temporally Correlated Latent Exploration for Reinforcement Learning","date":"2024-12-06","arxiv_id":"2412.04775","repositories_listed":0,"syntology":null},{"url":null,"slug":"putting-the-iterative-training-of-decision","title":"Putting the Iterative Training of Decision Trees to the Test on a Real-World Robotic Task","date":"2024-12-06","arxiv_id":"2412.04974","repositories_listed":0,"syntology":null},{"url":null,"slug":"demonstration-selection-for-in-context","title":"Demonstration Selection for In-Context Learning via Reinforcement Learning","date":"2024-12-05","arxiv_id":"2412.03966","repositories_listed":0,"syntology":null},{"url":null,"slug":"element-episodic-and-lifelong-exploration-via","title":"ELEMENT: Episodic and Lifelong Exploration via Maximum Entropy","date":"2024-12-05","arxiv_id":"2412.03800","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-freeway-lane","title":"Reinforcement Learning for Freeway Lane-Change Regulation via Connected Vehicles","date":"2024-12-05","arxiv_id":"2412.04341","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-from-wild-animal","title":"Reinforcement Learning from Wild Animal Videos","date":"2024-12-05","arxiv_id":"2412.04273","repositories_listed":0,"syntology":null},{"url":null,"slug":"traffic-co-simulation-framework-empowered-by","title":"Traffic Co-Simulation Framework Empowered by Infrastructure Camera Sensing and Reinforcement Learning","date":"2024-12-05","arxiv_id":"2412.03925","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-acquisition-for-improving-model-fairness","title":"Data Acquisition for Improving Model Fairness using Reinforcement Learning","date":"2024-12-04","arxiv_id":"2412.03009","repositories_listed":0,"syntology":null},{"url":null,"slug":"experience-driven-discovery-of-planning","title":"Experience-driven discovery of planning strategies","date":"2024-12-04","arxiv_id":"2412.03111","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyper-hyperparameter-robust-efficient","title":"Hyper: Hyperparameter Robust Efficient Exploration in Reinforcement Learning","date":"2024-12-04","arxiv_id":"2412.03767","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-delayed-reinforcement-learning","title":"Inverse Delayed Reinforcement Learning","date":"2024-12-04","arxiv_id":"2412.02931","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-driven-resource-allocation-framework-for","title":"AI-Driven Resource Allocation Framework for Microservices in Hybrid Cloud Platforms","date":"2024-12-03","arxiv_id":"2412.02610","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-problem-of-social-cost-in-multi-agent","title":"The Problem of Social Cost in Multi-Agent General Reinforcement Learning: Survey and Synthesis","date":"2024-12-03","arxiv_id":"2412.02091","repositories_listed":0,"syntology":null},{"url":null,"slug":"explore-reinforced-equilibrium-approximation","title":"Explore Reinforced: Equilibrium Approximation with Reinforcement Learning","date":"2024-12-02","arxiv_id":"2412.02016","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-adaptation-of-reinforcement-learning","title":"Task Adaptation of Reinforcement Learning-based NAS Agents through Transfer Learning","date":"2024-12-02","arxiv_id":"2412.01420","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-type-agnostic-cyber-defense-agents","title":"Towards Type Agnostic Cyber Defense Agents","date":"2024-12-02","arxiv_id":"2412.01542","repositories_listed":0,"syntology":null},{"url":null,"slug":"bilinear-convolution-decomposition-for-causal","title":"Bilinear Convolution Decomposition for Causal RL Interpretability","date":"2024-12-01","arxiv_id":"2412.00944","repositories_listed":0,"syntology":null},{"url":null,"slug":"mean-field-sampling-for-cooperative-multi","title":"Mean-Field Sampling for Cooperative Multi-Agent Reinforcement Learning","date":"2024-12-01","arxiv_id":"2412.00661","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-poisoning-attack-against-reinforcement","title":"Online Poisoning Attack Against Reinforcement Learning under Black-box Environments","date":"2024-12-01","arxiv_id":"2412.00797","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-partially-observable-reinforcement","title":"Provable Partially Observable Reinforcement Learning with Privileged Information","date":"2024-12-01","arxiv_id":"2412.00985","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-generalization-of-robot-locomotion","title":"Improving generalization of robot locomotion policies via Sharpness-Aware Reinforcement Learning","date":"2024-11-29","arxiv_id":"2411.19732","repositories_listed":0,"syntology":null},{"url":null,"slug":"proto-successor-measure-representing-the","title":"Proto Successor Measure: Representing the Space of All Possible Solutions of Reinforcement Learning","date":"2024-11-29","arxiv_id":"2411.19418","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-survey-of-reinforcement","title":"A Comprehensive Survey of Reinforcement Learning: From Algorithms to Practical Challenges","date":"2024-11-28","arxiv_id":"2411.18892","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-of-soft-actor-critic-algorithms","title":"Application of Soft Actor-Critic Algorithms in Optimizing Wastewater Treatment with Time Delays Integration","date":"2024-11-27","arxiv_id":"2411.18305","repositories_listed":0,"syntology":null},{"url":null,"slug":"dependency-aware-cav-task-scheduling-via","title":"Dependency-Aware CAV Task Scheduling via Diffusion-Based Reinforcement Learning","date":"2024-11-27","arxiv_id":"2411.18230","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-non-prehensile-object-transport-via","title":"Dynamic Non-Prehensile Object Transport via Model-Predictive Reinforcement Learning","date":"2024-11-27","arxiv_id":"2412.00086","repositories_listed":0,"syntology":null},{"url":null,"slug":"elemental-interactive-learning-from","title":"ELEMENTAL: Interactive Learning from Demonstrations and Vision-Language Models for Reward Design in Robotics","date":"2024-11-27","arxiv_id":"2411.18825","repositories_listed":0,"syntology":null},{"url":null,"slug":"bpp-search-enhancing-tree-of-thought","title":"BPP-Search: Enhancing Tree of Thought Reasoning for Mathematical Modeling Problem Solving","date":"2024-11-26","arxiv_id":"2411.17404","repositories_listed":0,"syntology":null},{"url":null,"slug":"crash-challenging-reinforcement-learning","title":"CRASH: Challenging Reinforcement-Learning Based Adversarial Scenarios For Safety Hardening","date":"2024-11-26","arxiv_id":"2411.16996","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensuring-safety-in-target-pursuit-control-a","title":"Ensuring Safety in Target Pursuit Control: A CBF-Safe Reinforcement Learning Approach","date":"2024-11-26","arxiv_id":"2411.17552","repositories_listed":0,"syntology":null},{"url":null,"slug":"broad-critic-deep-actor-reinforcement","title":"Broad Critic Deep Actor Reinforcement Learning for Continuous Control","date":"2024-11-24","arxiv_id":"2411.15806","repositories_listed":0,"syntology":null},{"url":null,"slug":"partial-identifiability-and-misspecification","title":"Partial Identifiability and Misspecification in Inverse Reinforcement Learning","date":"2024-11-24","arxiv_id":"2411.15951","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-enhanced-genetic","title":"Reinforcement learning-enhanced genetic algorithm for wind farm layout optimization","date":"2024-11-24","arxiv_id":"2412.06803","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-molecular-design-through-graph","title":"Enhancing Molecular Design through Graph-based Topological Reinforcement Learning","date":"2024-11-22","arxiv_id":"2411.14726","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-multi-agent-inverse-reinforcement-learning","title":"On Multi-Agent Inverse Reinforcement Learning","date":"2024-11-22","arxiv_id":"2411.15046","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-prediction-models-with","title":"Enhancing Prediction Models with Reinforcement Learning","date":"2024-11-21","arxiv_id":"2412.06791","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-action-manipulation-attack","title":"Provably Efficient Action-Manipulation Attack Against Continuous Reinforcement Learning","date":"2024-11-20","arxiv_id":"2411.13116","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-collusion-and-the-folk","title":"Reinforcement Learning, Collusion, and the Folk Theorem","date":"2024-11-19","arxiv_id":"2411.12725","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-action-sequence","title":"Coarse-to-fine Q-Network with Action Sequence for Data-Efficient Robot Learning","date":"2024-11-19","arxiv_id":"2411.12155","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-and-optimization-of-multi-rendezvous","title":"Design And Optimization Of Multi-rendezvous Manoeuvres Based On Reinforcement Learning And Convex Optimization","date":"2024-11-18","arxiv_id":"2411.11778","repositories_listed":0,"syntology":null},{"url":null,"slug":"no-regret-exploration-in-shuffle-private","title":"No-regret Exploration in Shuffle Private Reinforcement Learning","date":"2024-11-18","arxiv_id":"2411.11647","repositories_listed":0,"syntology":null},{"url":null,"slug":"preserving-expert-level-privacy-in-offline","title":"Preserving Expert-Level Privacy in Offline Reinforcement Learning","date":"2024-11-18","arxiv_id":"2411.13598","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-free-reinforcement-learning-for-ltl","title":"Regret-Free Reinforcement Learning for LTL Specifications","date":"2024-11-18","arxiv_id":"2411.12019","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-under-diffusion","title":"Robust Reinforcement Learning under Diffusion Models for Data with Jumps","date":"2024-11-18","arxiv_id":"2411.11697","repositories_listed":0,"syntology":null},{"url":null,"slug":"theoretical-corrections-and-the-leveraging-of","title":"Theoretical Corrections and the Leveraging of Reinforcement Learning to Enhance Triangle Attack","date":"2024-11-18","arxiv_id":"2411.12071","repositories_listed":0,"syntology":null},{"url":null,"slug":"upside-down-reinforcement-learning-for-more","title":"Upside-Down Reinforcement Learning for More Interpretable Optimal Control","date":"2024-11-18","arxiv_id":"2411.11457","repositories_listed":0,"syntology":null},{"url":null,"slug":"financial-news-driven-llm-reinforcement","title":"Financial News-Driven LLM Reinforcement Learning for Portfolio Management","date":"2024-11-17","arxiv_id":"2411.11059","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-relative-over-generalization-in","title":"Mitigating Relative Over-Generalization in Multi-Agent Reinforcement Learning","date":"2024-11-17","arxiv_id":"2411.11099","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-low-regret-online-reinforcement","title":"Efficient, Low-Regret, Online Reinforcement Learning for Linear MDPs","date":"2024-11-16","arxiv_id":"2411.10906","repositories_listed":0,"syntology":null},{"url":null,"slug":"stable-continual-reinforcement-learning-via","title":"Stable Continual Reinforcement Learning via Diffusion-based Trajectory Replay","date":"2024-11-16","arxiv_id":"2411.10809","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-risk-sensitive-contract-unified","title":"A Risk Sensitive Contract-unified Reinforcement Learning Approach for Option Hedging","date":"2024-11-14","arxiv_id":"2411.09659","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-reinforcement-learning-for","title":"Enhancing reinforcement learning for population setpoint tracking in co-cultures","date":"2024-11-14","arxiv_id":"2411.09177","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-batch-reinforcement-learning-via","title":"Iterative Batch Reinforcement Learning via Safe Diversified Model-based Policy Search","date":"2024-11-14","arxiv_id":"2411.09722","repositories_listed":0,"syntology":null},{"url":null,"slug":"rationality-based-innate-values-driven","title":"Innate-Values-driven Reinforcement Learning based Cognitive Modeling","date":"2024-11-14","arxiv_id":"2411.09160","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-disentanglers-on-random-unitary","title":"Reinforced Disentanglers on Random Unitary Circuits","date":"2024-11-14","arxiv_id":"2411.09784","repositories_listed":0,"syntology":null},{"url":null,"slug":"bamax-backtrack-assisted-multi-agent","title":"BAMAX: Backtrack Assisted Multi-Agent Exploration using Reinforcement Learning","date":"2024-11-13","arxiv_id":"2411.08400","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-mixed-llm-improve-large-language-models","title":"Code-mixed LLM: Improve Large Language Models' Capability to Handle Code-Mixing through Reinforcement Learning from AI Feedback","date":"2024-11-13","arxiv_id":"2411.09073","repositories_listed":0,"syntology":null},{"url":null,"slug":"liner-shipping-network-design-with","title":"Liner Shipping Network Design with Reinforcement Learning","date":"2024-11-13","arxiv_id":"2411.09068","repositories_listed":0,"syntology":null},{"url":null,"slug":"r3hf-reward-redistribution-for-enhancing","title":"R3HF: Reward Redistribution for Enhancing Reinforcement Learning from Human Feedback","date":"2024-11-13","arxiv_id":"2411.08302","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlinspect-an-interactive-visual-approach-to","title":"RLInspect: An Interactive Visual Approach to Assess Reinforcement Learning Algorithm","date":"2024-11-13","arxiv_id":"2411.08392","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-multi-agent-reinforcement-learning","title":"Exploring Multi-Agent Reinforcement Learning for Unrelated Parallel Machine Scheduling","date":"2024-11-12","arxiv_id":"2411.07634","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-framework-for-1","title":"Reinforcement Learning Framework for Quantitative Trading","date":"2024-11-12","arxiv_id":"2411.07585","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-offline-reinforcement-learning-for-non","title":"Robust Offline Reinforcement Learning for Non-Markovian Decision Processes","date":"2024-11-12","arxiv_id":"2411.07514","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-differential-patient-care-through","title":"Identifying Differential Patient Care Through Inverse Intent Inference","date":"2024-11-11","arxiv_id":"2411.07372","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-hop-upstream-preemptive-traffic-signal","title":"Multi-hop Upstream Anticipatory Traffic Signal Control with Deep Reinforcement Learning","date":"2024-11-10","arxiv_id":"2411.07271","repositories_listed":0,"syntology":null},{"url":null,"slug":"offlight-an-offline-multi-agent-reinforcement","title":"OffLight: An Offline Multi-Agent Reinforcement Learning Framework for Traffic Signal Control","date":"2024-11-10","arxiv_id":"2411.06601","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-execution-with-reinforcement-learning","title":"Optimal Execution with Reinforcement Learning","date":"2024-11-10","arxiv_id":"2411.06389","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-digital-twin","title":"Deep Reinforcement Learning for Digital Twin-Oriented Complex Networked Systems","date":"2024-11-09","arxiv_id":"2411.06148","repositories_listed":0,"syntology":null},{"url":null,"slug":"acceleration-for-deep-reinforcement-learning","title":"Acceleration for Deep Reinforcement Learning using Parallel and Distributed Computing: A Survey","date":"2024-11-08","arxiv_id":"2411.05614","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-offline-reinforcement-learning-1","title":"Real-World Offline Reinforcement Learning from Vision Language Model Feedback","date":"2024-11-08","arxiv_id":"2411.05273","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-adaptive-resource","title":"Reinforcement Learning for Adaptive Resource Scheduling in Complex System Environments","date":"2024-11-08","arxiv_id":"2411.05346","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-active-flow-control-strategies","title":"Towards Active Flow Control Strategies Through Deep Reinforcement Learning","date":"2024-11-08","arxiv_id":"2411.05536","repositories_listed":0,"syntology":null},{"url":null,"slug":"tract-rlformer-a-tract-specific-rl-policy","title":"Tract-RLFormer: A Tract-Specific RL policy based Decoder-only Transformer Network","date":"2024-11-08","arxiv_id":"2411.05757","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-robustness-of-reinforcement","title":"Evaluating Robustness of Reinforcement Learning Algorithms for Autonomous Shipping","date":"2024-11-07","arxiv_id":"2411.04915","repositories_listed":0,"syntology":null},{"url":null,"slug":"harnessing-the-power-of-gradient-based","title":"Harnessing the Power of Gradient-Based Simulations for Multi-Objective Optimization in Particle Accelerators","date":"2024-11-07","arxiv_id":"2411.04817","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-dialogue-agents-via-reinforcement","title":"Interactive Dialogue Agents via Reinforcement Learning on Hindsight Regenerations","date":"2024-11-07","arxiv_id":"2411.05194","repositories_listed":0,"syntology":null},{"url":null,"slug":"performative-reinforcement-learning-with","title":"Performative Reinforcement Learning with Linear Markov Decision Process","date":"2024-11-07","arxiv_id":"2411.05234","repositories_listed":0,"syntology":null}],"record_sha256":"869aabf37bbed7f8fb65b42d4b0c5c341f50f9eb0517c18f1d5f271a20f2e7b4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}