{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/67","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":67,"pages_in_order":152,"rows_per_page":100,"rows":[6601,6700],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/66","next":"/task/reinforcement-learning-1/papers/68","papers":[{"url":null,"slug":"digital-twin-native-ai-driven-service","title":"Digital Twin-Native AI-Driven Service Architecture for Industrial Networks","date":"2023-11-24","arxiv_id":"2311.14532","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-pretrained-models-for-deployable","title":"Evaluating Pretrained models for Deployable Lifelong Learning","date":"2023-11-22","arxiv_id":"2311.13648","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-inference-in-reinforcement","title":"Probabilistic Inference in Reinforcement Learning Done Right","date":"2023-11-22","arxiv_id":"2311.13294","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-markov-decision-process-and","title":"Risk-sensitive Markov Decision Process and Learning under General Utility Functions","date":"2023-11-22","arxiv_id":"2311.13589","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-behaviors-of-mixed-traffic-via","title":"Analyzing Behaviors of Mixed Traffic via Reinforcement Learning at Unsignalized Intersections","date":"2023-11-21","arxiv_id":"2312.05325","repositories_listed":0,"syntology":null},{"url":null,"slug":"clustered-policy-decision-ranking","title":"Clustered Policy Decision Ranking","date":"2023-11-21","arxiv_id":"2311.12970","repositories_listed":0,"syntology":null},{"url":null,"slug":"resilient-control-of-networked-microgrids","title":"Resilient Control of Networked Microgrids using Vertical Federated Reinforcement Learning: Designs and Real-Time Test-Bed Validations","date":"2023-11-21","arxiv_id":"2311.12264","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-cvar-rl-in-low-rank-mdps","title":"Provably Efficient CVaR RL in Low-rank MDPs","date":"2023-11-20","arxiv_id":"2311.11965","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-and-deep-stochastic","title":"Reinforcement Learning and Deep Stochastic Optimal Control for Final Quadratic Hedging","date":"2023-11-20","arxiv_id":"2401.08600","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-for-wireless","title":"Offline Reinforcement Learning for Wireless Network Optimization with Mixture Datasets","date":"2023-11-19","arxiv_id":"2311.11423","repositories_listed":0,"syntology":null},{"url":null,"slug":"tactile-active-inference-reinforcement","title":"Tactile Active Inference Reinforcement Learning for Efficient Robotic Manipulation Skill Acquisition","date":"2023-11-19","arxiv_id":"2311.11287","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-feature-extractors-for","title":"Benchmarking Feature Extractors for Reinforcement Learning-Based Semiconductor Defect Localization","date":"2023-11-18","arxiv_id":"2311.11145","repositories_listed":0,"syntology":null},{"url":null,"slug":"imagination-augmented-hierarchical","title":"Imagination-Augmented Hierarchical Reinforcement Learning for Safe and Interactive Autonomous Driving in Urban Environments","date":"2023-11-17","arxiv_id":"2311.10309","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-lqr-using-reinforcement-learning","title":"Data-Driven LQR using Reinforcement Learning and Quadratic Neural Networks","date":"2023-11-16","arxiv_id":"2311.10235","repositories_listed":0,"syntology":null},{"url":null,"slug":"runtime-verification-of-learning-properties","title":"Runtime Verification of Learning Properties for Reinforcement Learning Algorithms","date":"2023-11-16","arxiv_id":"2311.09811","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-imitation-learning-on-aggregated","title":"Adversarial Imitation Learning On Aggregated Data","date":"2023-11-14","arxiv_id":"2311.08568","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-policy-policy-gradient-reinforcement","title":"On-Policy Policy Gradient Reinforcement Learning Without On-Policy Sampling","date":"2023-11-14","arxiv_id":"2311.08290","repositories_listed":0,"syntology":null},{"url":null,"slug":"purpose-in-the-machine-do-traffic-simulators","title":"Purpose in the Machine: Do Traffic Simulators Produce Distributionally Equivalent Outcomes for Reinforcement Learning Applications?","date":"2023-11-14","arxiv_id":"2311.08429","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-mining-electric-locomotives-meet","title":"When Mining Electric Locomotives Meet Reinforcement Learning","date":"2023-11-14","arxiv_id":"2311.08153","repositories_listed":0,"syntology":null},{"url":null,"slug":"workflow-guided-response-generation-for-task","title":"Workflow-Guided Response Generation for Task-Oriented Dialogue","date":"2023-11-14","arxiv_id":"2311.08300","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-introduction-to-reinforcement-learning-for","title":"An introduction to reinforcement learning for neuroscience","date":"2023-11-13","arxiv_id":"2311.07315","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-robustness-in-cyber-physical","title":"Investigating Robustness in Cyber-Physical Systems: Specification-Centric Analysis in the face of System Deviations","date":"2023-11-13","arxiv_id":"2311.07462","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-advantage-based-policy-transfer-algorithm","title":"An advantage based policy transfer algorithm for reinforcement learning with measures of transferability","date":"2023-11-12","arxiv_id":"2311.06731","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-predictive-safety-filter-via","title":"Learning Predictive Safety Filter via Decomposition of Robust Invariant Set","date":"2023-11-12","arxiv_id":"2311.06769","repositories_listed":0,"syntology":null},{"url":null,"slug":"genetic-algorithm-enhanced-by-deep","title":"Genetic Algorithm enhanced by Deep Reinforcement Learning in parent selection mechanism and mutation : Minimizing makespan in permutation flow shop scheduling problems","date":"2023-11-10","arxiv_id":"2311.05937","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-augmented-scheduling-for-solar","title":"Out-of-Distribution-Aware Electric Vehicle Charging","date":"2023-11-10","arxiv_id":"2311.05941","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-what-to-when-a-spiking-neural-network","title":"From \"What\" to \"When\" -- a Spiking Neural Network Predicting Rare Events and Time to their Occurrence","date":"2023-11-09","arxiv_id":"2311.05210","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-stochastic-nonlinear-model","title":"Adaptive Stochastic Nonlinear Model Predictive Control with Look-ahead Deep Reinforcement Learning for Autonomous Vehicle Motion Control","date":"2023-11-07","arxiv_id":"2311.04303","repositories_listed":0,"syntology":null},{"url":null,"slug":"stable-modular-control-via-contraction-theory","title":"Stable Modular Control via Contraction Theory for Reinforcement Learning","date":"2023-11-07","arxiv_id":"2311.03669","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-rank-mdps-with-continuous-action-spaces","title":"Low-Rank MDPs with Continuous Action Spaces","date":"2023-11-06","arxiv_id":"2311.03564","repositories_listed":0,"syntology":null},{"url":null,"slug":"virtual-action-actor-critic-framework-for","title":"Virtual Action Actor-Critic Framework for Exploration (Student Abstract)","date":"2023-11-06","arxiv_id":"2311.02916","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-dimensional-bid-learning-for-energy","title":"High-dimensional Bid Learning for Energy Storage Bidding in Energy Markets","date":"2023-11-05","arxiv_id":"2311.02551","repositories_listed":0,"syntology":null},{"url":null,"slug":"pointer-networks-with-q-learning-for-op","title":"Pointer Networks with Q-Learning for Combinatorial Optimization","date":"2023-11-05","arxiv_id":"2311.02629","repositories_listed":0,"syntology":null},{"url":null,"slug":"staged-reinforcement-learning-for-complex","title":"Staged Reinforcement Learning for Complex Tasks through Decomposed Environments","date":"2023-11-05","arxiv_id":"2311.02746","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-reinforcement-learning-of","title":"Accelerating Reinforcement Learning of Robotic Manipulations via Feedback from Large Language Models","date":"2023-11-04","arxiv_id":"2311.02379","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-randomization-via-entropy-maximization","title":"Domain Randomization via Entropy Maximization","date":"2023-11-03","arxiv_id":"2311.01885","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficiency-optimization-for-1","title":"Energy Efficiency Optimization for Subterranean LoRaWAN Using A Reinforcement Learning Approach: A Direct-to-Satellite Scenario","date":"2023-11-03","arxiv_id":"2311.01743","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-bootstrapped-reinforcement-learning","title":"Imitation Bootstrapped Reinforcement Learning","date":"2023-11-03","arxiv_id":"2311.02198","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-adversarial-reinforcement-learning-via","title":"Robust Adversarial Reinforcement Learning via Bounded Rationality Curricula","date":"2023-11-03","arxiv_id":"2311.01642","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-model-free-rl-algorithms-that-scale","title":"Towards model-free RL algorithms that scale well with unstructured data","date":"2023-11-03","arxiv_id":"2311.02215","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-general-value-functions-to-learn-domain","title":"Using General Value Functions to Learn Domain-Backed Inventory Management Policies","date":"2023-11-03","arxiv_id":"2311.02125","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-information-propagation-in","title":"Analysis of Information Propagation in Ethereum Network Using Combined Graph Attention Network and Reinforcement Learning to Optimize Network Efficiency and Scalability","date":"2023-11-02","arxiv_id":"2311.01406","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-realistic-traffic-agents-in-closed","title":"Learning Realistic Traffic Agents in Closed-loop","date":"2023-11-02","arxiv_id":"2311.01394","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-generalization-through","title":"Enhanced Generalization through Prioritization and Diversity in Self-Imitation Reinforcement Learning over Procedural Environments with Sparse Rewards","date":"2023-11-01","arxiv_id":"2311.00426","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-natural-policy-gradient-methods-for","title":"Federated Natural Policy Gradient and Actor Critic Methods for Multi-task Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00201","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-impartial-policies-for-sequential","title":"Learning impartial policies for sequential counterfactual explanations using Deep Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00523","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-decision-transformer-via","title":"Rethinking Decision Transformer via Hierarchical Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00267","repositories_listed":0,"syntology":null},{"url":null,"slug":"expressive-modeling-is-insufficient-for","title":"A Tractable Inference Perspective of Offline RL","date":"2023-10-31","arxiv_id":"2311.00094","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-rl-with-observation-histories","title":"Offline RL with Observation Histories: Analyzing and Improving Sample Complexity","date":"2023-10-31","arxiv_id":"2310.20663","repositories_listed":0,"syntology":null},{"url":null,"slug":"pick-and-pass-as-a-hat-trick-class-for-first","title":"Closed Drafting as a Case Study for First-Principle Interpretability, Memory, and Generalizability in Deep Reinforcement Learning","date":"2023-10-31","arxiv_id":"2310.20654","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-aware-causal-representation-for","title":"Safety-aware Causal Representation for Trustworthy Offline Reinforcement Learning in Autonomous Driving","date":"2023-10-31","arxiv_id":"2311.10747","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-and-safe-deep-reinforcement","title":"Sample-Efficient and Safe Deep Reinforcement Learning via Reset Deep Ensemble Agents","date":"2023-10-31","arxiv_id":"2310.20287","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoupled-actor-critic","title":"On the Theory of Risk-Aware Agents: Bridging Actor-Critic and Economics","date":"2023-10-30","arxiv_id":"2310.19527","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversify-conquer-outcome-directed-curriculum","title":"Diversify & Conquer: Outcome-directed Curriculum RL via Out-of-Distribution Disagreement","date":"2023-10-30","arxiv_id":"2310.19261","repositories_listed":0,"syntology":null},{"url":null,"slug":"posterior-sampling-for-competitive-rl","title":"Posterior Sampling for Competitive RL: Function Approximation and Partial Observation","date":"2023-10-30","arxiv_id":"2310.19861","repositories_listed":0,"syntology":null},{"url":null,"slug":"automaton-distillation-neuro-symbolic","title":"Automaton Distillation: Neuro-Symbolic Transfer Learning for Deep Reinforcement Learning","date":"2023-10-29","arxiv_id":"2310.19137","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-alignment-via-reward-function","title":"Behavior Alignment via Reward Function Optimization","date":"2023-10-29","arxiv_id":"2310.19007","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-agents-with-reinforcement-learning","title":"Language Agents with Reinforcement Learning for Strategic Play in the Werewolf Game","date":"2023-10-29","arxiv_id":"2310.18940","repositories_listed":0,"syntology":null},{"url":null,"slug":"mag-gnn-reinforcement-learning-boosted-graph","title":"MAG-GNN: Reinforcement Learning Boosted Graph Neural Network","date":"2023-10-29","arxiv_id":"2310.19142","repositories_listed":0,"syntology":null},{"url":null,"slug":"posterior-sampling-with-delayed-feedback-for","title":"Posterior Sampling with Delayed Feedback for Reinforcement Learning with Linear Function Approximation","date":"2023-10-29","arxiv_id":"2310.18919","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-implementation-of-reinforcement","title":"Real-World Implementation of Reinforcement Learning Based Energy Coordination for a Cluster of Households","date":"2023-10-29","arxiv_id":"2310.19155","repositories_listed":0,"syntology":null},{"url":null,"slug":"spacecraft-autonomous-decision-planning-for","title":"Spacecraft Autonomous Decision-Planning for Collision Avoidance: a Reinforcement Learning Approach","date":"2023-10-29","arxiv_id":"2310.18966","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-behavior-extraction-via-random","title":"Unsupervised Behavior Extraction via Random Intent Priors","date":"2023-10-28","arxiv_id":"2310.18687","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-weapons-to","title":"Deep Reinforcement Learning for Weapons to Targets Assignment in a Hypersonic strike","date":"2023-10-27","arxiv_id":"2310.18509","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-data-augmentation-for-offline","title":"Guided Data Augmentation for Offline Reinforcement Learning and Imitation Learning","date":"2023-10-27","arxiv_id":"2310.18247","repositories_listed":0,"syntology":null},{"url":null,"slug":"cqm-curriculum-reinforcement-learning-with-a","title":"CQM: Curriculum Reinforcement Learning with a Quantized World Model","date":"2023-10-26","arxiv_id":"2310.17330","repositories_listed":0,"syntology":null},{"url":null,"slug":"demonstration-regularized-rl","title":"Demonstration-Regularized RL","date":"2023-10-26","arxiv_id":"2310.17303","repositories_listed":0,"syntology":null},{"url":null,"slug":"grow-your-limits-continuous-improvement-with","title":"Grow Your Limits: Continuous Improvement with Real-World RL for Robotic Locomotion","date":"2023-10-26","arxiv_id":"2310.17634","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlled-decoding-from-language-models","title":"Controlled Decoding from Language Models","date":"2023-10-25","arxiv_id":"2310.17022","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-enhanced-contrastive-reinforcement","title":"Model-enhanced Contrastive Reinforcement Learning for Sequential Recommendation","date":"2023-10-25","arxiv_id":"2310.16566","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiprompter-cooperative-prompt-optimization","title":"MultiPrompter: Cooperative Prompt Optimization with Multi-Agent Reinforcement Learning","date":"2023-10-25","arxiv_id":"2310.16730","repositories_listed":0,"syntology":null},{"url":null,"slug":"privately-aligning-language-models-with","title":"Privately Aligning Language Models with Reinforcement Learning","date":"2023-10-25","arxiv_id":"2310.16960","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-of-reinforcement-learning-based","title":"Transfer of Reinforcement Learning-Based Controllers from Model- to Hardware-in-the-Loop","date":"2023-10-25","arxiv_id":"2310.17671","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-contextualized-real-time-multimodal-emotion","title":"A Contextualized Real-Time Multimodal Emotion Recognition for Conversational Agents using Graph Convolutional Networks in Reinforcement Learning","date":"2023-10-24","arxiv_id":"2310.18363","repositories_listed":0,"syntology":null},{"url":null,"slug":"finetuning-offline-world-models-in-the-real","title":"Finetuning Offline World Models in the Real World","date":"2023-10-24","arxiv_id":"2310.16029","repositories_listed":0,"syntology":null},{"url":null,"slug":"fractal-landscapes-in-policy-optimization","title":"Fractal Landscapes in Policy Optimization","date":"2023-10-24","arxiv_id":"2310.15418","repositories_listed":0,"syntology":null},{"url":null,"slug":"webwise-web-interface-control-and-sequential","title":"WebWISE: Web Interface Control and Sequential Exploration with Large Language Models","date":"2023-10-24","arxiv_id":"2310.16042","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-reinforcement-learning-for-1","title":"A Review of Reinforcement Learning for Natural Language Processing, and Applications in Healthcare","date":"2023-10-23","arxiv_id":"2310.18354","repositories_listed":0,"syntology":null},{"url":null,"slug":"diverse-priors-for-deep-reinforcement","title":"Diverse Priors for Deep Reinforcement Learning","date":"2023-10-23","arxiv_id":"2310.14864","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-robotic-manipulation-harnessing-the","title":"Enhancing Robotic Manipulation: Harnessing the Power of Multi-Task Reinforcement Learning and Single Life Reinforcement Learning in Meta-World","date":"2023-10-23","arxiv_id":"2311.12854","repositories_listed":0,"syntology":null},{"url":null,"slug":"iteratively-learn-diverse-strategies-with","title":"Iteratively Learn Diverse Strategies with State Distance Information","date":"2023-10-23","arxiv_id":"2310.14509","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-large-structured","title":"Reinforcement learning in large, structured action spaces: A simulation study of decision support for spinal cord injury rehabilitation","date":"2023-10-23","arxiv_id":"2310.14976","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-benefits-of-multi-task-rl-under-non","title":"Provable Benefits of Multi-task RL under Non-Markovian Decision Making Processes","date":"2023-10-20","arxiv_id":"2310.13550","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerate-presolve-in-large-scale-linear","title":"Accelerate Presolve in Large-Scale Linear Programming via Reinforcement Learning","date":"2023-10-18","arxiv_id":"2310.11845","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-quantized-offline-reinforcement","title":"Action-Quantized Offline Reinforcement Learning for Robotic Skill Learning","date":"2023-10-18","arxiv_id":"2310.11731","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-generalization-of-alignment-with","title":"Improving Generalization of Alignment with Human Preferences through Group Invariant Learning","date":"2023-10-18","arxiv_id":"2310.11971","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-solve-climate-sensor-placement","title":"Learning to Optimise Climate Sensor Placement using a Transformer","date":"2023-10-18","arxiv_id":"2310.12387","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-expressivity-of-objective","title":"On The Expressivity of Objective-Specification Formalisms in Reinforcement Learning","date":"2023-10-18","arxiv_id":"2310.11840","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-experience-classification-for-training","title":"Using Experience Classification for Training Non-Markovian Tasks","date":"2023-10-18","arxiv_id":"2310.11678","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-packing-from-visual-sensing-to","title":"Neural Packing: from Visual Sensing to Reinforcement Learning","date":"2023-10-17","arxiv_id":"2311.09233","repositories_listed":0,"syntology":null},{"url":null,"slug":"reaching-the-limit-in-autonomous-racing","title":"Reaching the Limit in Autonomous Racing: Optimal Control versus Reinforcement Learning","date":"2023-10-17","arxiv_id":"2310.10943","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-topological-maps-in-deep","title":"Leveraging Topological Maps in Deep Reinforcement Learning for Multi-Object Navigation","date":"2023-10-16","arxiv_id":"2310.10250","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-transfer-across-tasks-using","title":"Uncertainty-aware transfer across tasks using hybrid model-based successor feature reinforcement learning","date":"2023-10-16","arxiv_id":"2310.10818","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-explicit","title":"Deep Reinforcement Learning with Explicit Context Representation","date":"2023-10-15","arxiv_id":"2310.09924","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-empowering-reinforcement","title":"A Framework for Empowering Reinforcement Learning Agents with Causal Analysis: Enhancing Automated Cryptocurrency Trading","date":"2023-10-14","arxiv_id":"2310.09462","repositories_listed":0,"syntology":null},{"url":null,"slug":"lgts-dynamic-task-sampling-using-llm","title":"LgTS: Dynamic Task Sampling using LLM-generated sub-goals for Reinforcement Learning Agents","date":"2023-10-14","arxiv_id":"2310.09454","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-with-principles-for-diverse-ai","title":"Exploration with Principles for Diverse AI Supervision","date":"2023-10-13","arxiv_id":"2310.08899","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-reinforcement-learning-for-optimizing","title":"Hybrid Reinforcement Learning for Optimizing Pump Sustainability in Real-World Water Distribution Networks","date":"2023-10-13","arxiv_id":"2310.09412","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-optimal-transport-for-enhanced","title":"Leveraging Optimal Transport for Enhanced Offline Reinforcement Learning in Surgical Robotic Environments","date":"2023-10-13","arxiv_id":"2310.08841","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lightweight-calibrated-simulation-enabling","title":"A Lightweight Calibrated Simulation Enabling Efficient Offline Learning for Optimal Control of Real Buildings","date":"2023-10-12","arxiv_id":"2310.08569","repositories_listed":0,"syntology":null}],"record_sha256":"5c5b1e46b7c4ef7dcad5f7f9df820d545b680d64fd00905816d070e615efd14c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}