{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/61","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":61,"pages_in_order":132,"rows_per_page":100,"rows":[6001,6100],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/60","next":"/task/reinforcement-learning/papers/62","papers":[{"url":null,"slug":"exploiting-causal-graph-priors-with-posterior","title":"Exploiting Causal Graph Priors with Posterior Sampling for Reinforcement Learning","date":"2023-10-11","arxiv_id":"2310.07518","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-knowledge-graph","title":"Reinforcement Learning-based Knowledge Graph Reasoning for Explainable Fact-checking","date":"2023-10-11","arxiv_id":"2310.07613","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-safe-reinforcement-learning-under","title":"Robust Safe Reinforcement Learning under Adversarial Disturbances","date":"2023-10-11","arxiv_id":"2310.07207","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-multi-level-optimization-over","title":"Federated Multi-Level Optimization over Decentralized Networks","date":"2023-10-10","arxiv_id":"2310.06217","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-content-exploration","title":"Information Content Exploration","date":"2023-10-10","arxiv_id":"2310.06777","repositories_listed":0,"syntology":null},{"url":null,"slug":"realizing-stabilized-landing-for-computation","title":"Realizing Stabilized Landing for Computation-Limited Reusable Rockets: A Quantum Reinforcement Learning Approach","date":"2023-10-10","arxiv_id":"2310.06541","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-a-safety-embedded","title":"Reinforcement Learning in a Safety-Embedded MDP with Trajectory Optimization","date":"2023-10-10","arxiv_id":"2310.06903","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-semantic-non-markovian-simulation","title":"Scalable Semantic Non-Markovian Simulation Proxy for Reinforcement Learning","date":"2023-10-10","arxiv_id":"2310.06835","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-entry-wise-matrix-estimation-for-low","title":"Spectral Entry-wise Matrix Estimation for Low-Rank Reinforcement Learning","date":"2023-10-10","arxiv_id":"2310.06793","repositories_listed":0,"syntology":null},{"url":null,"slug":"factual-and-personalized-recommendations","title":"Factual and Personalized Recommendations using Language Models and Reinforcement Learning","date":"2023-10-09","arxiv_id":"2310.06176","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-9","title":"Hierarchical Reinforcement Learning for Temporal Pattern Prediction","date":"2023-10-09","arxiv_id":"2310.05695","repositories_listed":0,"syntology":null},{"url":null,"slug":"logic-guided-deep-reinforcement-learning-for","title":"Logic-Q: Improving Deep Reinforcement Learning-based Quantitative Trading via Program Sketch-based Tuning","date":"2023-10-09","arxiv_id":"2310.05551","repositories_listed":0,"syntology":null},{"url":null,"slug":"molecular-de-novo-design-through-transformer","title":"Molecular De Novo Design through Transformer-based Reinforcement Learning","date":"2023-10-09","arxiv_id":"2310.05365","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-timestep-models-for-model-based","title":"Multi-timestep models for Model-based Reinforcement Learning","date":"2023-10-09","arxiv_id":"2310.05672","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-freeform-robot","title":"Reinforcement learning for freeform robot design","date":"2023-10-09","arxiv_id":"2310.05670","repositories_listed":0,"syntology":null},{"url":null,"slug":"replication-of-multi-agent-reinforcement","title":"Replication of Multi-agent Reinforcement Learning for the \"Hide and Seek\" Problem","date":"2023-10-09","arxiv_id":"2310.05430","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-consistent-dynamics-models-are","title":"Reward-Consistent Dynamics Models are Strongly Generalizable for Offline Reinforcement Learning","date":"2023-10-09","arxiv_id":"2310.05422","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-is-agnostic-reinforcement-learning","title":"When is Agnostic Reinforcement Learning Statistically Tractable?","date":"2023-10-09","arxiv_id":"2310.06113","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-with-7","title":"Distributional Reinforcement Learning with Online Risk-awareness Adaption","date":"2023-10-08","arxiv_id":"2310.05179","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-spiking-neural-network-for-legged","title":"Fully Spiking Neural Network for Legged Robots","date":"2023-10-08","arxiv_id":"2310.05022","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-sequential-decision-making-in","title":"Optimal Sequential Decision-Making in Geosteering: A Reinforcement Learning Approach","date":"2023-10-07","arxiv_id":"2310.04772","repositories_listed":0,"syntology":null},{"url":null,"slug":"adjustable-robust-reinforcement-learning-for","title":"Adjustable Robust Reinforcement Learning for Online 3D Bin Packing","date":"2023-10-06","arxiv_id":"2310.04323","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-reinforcement-learning-to-option","title":"Applying Reinforcement Learning to Option Pricing and Hedging","date":"2023-10-06","arxiv_id":"2310.04336","repositories_listed":0,"syntology":null},{"url":null,"slug":"deconstructing-cooperation-and-ostracism-via","title":"Deconstructing Cooperation and Ostracism via Multi-Agent Reinforcement Learning","date":"2023-10-06","arxiv_id":"2310.04623","repositories_listed":0,"syntology":null},{"url":null,"slug":"searching-for-optimal-runtime-assurance-via","title":"Searching for Optimal Runtime Assurance via Reachability and Reinforcement Learning","date":"2023-10-06","arxiv_id":"2310.04288","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-kernel-perspective-on-behavioural-metrics","title":"A Kernel Perspective on Behavioural Metrics for Markov Decision Processes","date":"2023-10-05","arxiv_id":"2310.19804","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-deep-reinforcement-learning-in","title":"A Review of Deep Reinforcement Learning in Serverless Computing: Function Scheduling and Resource Auto-Scaling","date":"2023-10-05","arxiv_id":"2311.12839","repositories_listed":0,"syntology":null},{"url":null,"slug":"constraint-conditioned-policy-optimization","title":"Constraint-Conditioned Policy Optimization for Versatile Safe Reinforcement Learning","date":"2023-10-05","arxiv_id":"2310.03718","repositories_listed":0,"syntology":null},{"url":null,"slug":"pydcm-custom-data-center-models-with","title":"PyDCM: Custom Data Center Models with Reinforcement Learning for Sustainability","date":"2023-10-05","arxiv_id":"2310.03906","repositories_listed":0,"syntology":null},{"url":null,"slug":"rtdk-bo-high-dimensional-bayesian","title":"RTDK-BO: High Dimensional Bayesian Optimization with Reinforced Transformer Deep kernels","date":"2023-10-05","arxiv_id":"2310.03912","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-exploration-in-reinforcement-learning-a","title":"Safe Exploration in Reinforcement Learning: A Generalized Formulation and Algorithms","date":"2023-10-05","arxiv_id":"2310.03225","repositories_listed":0,"syntology":null},{"url":null,"slug":"small-batch-deep-reinforcement-learning","title":"Small batch deep reinforcement learning","date":"2023-10-05","arxiv_id":"2310.03882","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-architecture-impact-on-identifying","title":"Neural architecture impact on identifying temporally extended Reinforcement Learning tasks","date":"2023-10-04","arxiv_id":"2310.03161","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-estimation-and-inference-for-robust","title":"Online Estimation and Inference for Robust Policy Evaluation in Reinforcement Learning","date":"2023-10-04","arxiv_id":"2310.02581","repositories_listed":0,"syntology":null},{"url":null,"slug":"proximal-policy-optimization-based-1","title":"Proximal Policy Optimization-Based Reinforcement Learning Approach for DC-DC Boost Converter Control: A Comparative Evaluation Against Traditional Control Techniques","date":"2023-10-04","arxiv_id":"2310.02945","repositories_listed":0,"syntology":null},{"url":null,"slug":"searching-for-high-value-molecules-using","title":"Searching for High-Value Molecules Using Reinforcement Learning and Transformers","date":"2023-10-04","arxiv_id":"2310.02902","repositories_listed":0,"syntology":null},{"url":null,"slug":"blending-imitation-and-reinforcement-learning","title":"Blending Imitation and Reinforcement Learning for Robust Policy Improvement","date":"2023-10-03","arxiv_id":"2310.01737","repositories_listed":0,"syntology":null},{"url":null,"slug":"pessimistic-nonlinear-least-squares-value","title":"Pessimistic Nonlinear Least-Squares Value Iteration for Offline Reinforcement Learning","date":"2023-10-02","arxiv_id":"2310.01380","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-inhibitory-control-for-safe","title":"Risk-Sensitive Inhibitory Control for Safe Reinforcement Learning","date":"2023-10-02","arxiv_id":"2310.01538","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficiency-in-multi-batch","title":"Sample-Efficiency in Multi-Batch Reinforcement Learning: The Need for Dimension-Dependent Adaptivity","date":"2023-10-02","arxiv_id":"2310.01616","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-the-quadratic-assignment-problem","title":"Solving the Quadratic Assignment Problem using Deep Reinforcement Learning","date":"2023-10-02","arxiv_id":"2310.01604","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-general-offline-reinforcement-learning","title":"A General Offline Reinforcement Learning Framework for Interactive Recommendation","date":"2023-10-01","arxiv_id":"2310.00678","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-bandits-model-to-deep-deterministic","title":"From Bandits Model to Deep Deterministic Policy Gradient, Reinforcement Learning with Contextual Information","date":"2023-10-01","arxiv_id":"2310.00642","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-quantum-system-control-method-based-on","title":"A quantum system control method based on enhanced reinforcement learning","date":"2023-09-30","arxiv_id":"2310.03036","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-neural-style-transfer-with-deep","title":"Controlling Neural Style Transfer with Deep Reinforcement Learning","date":"2023-09-30","arxiv_id":"2310.00405","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-autonomous-6","title":"Deep Reinforcement Learning for Autonomous Vehicle Intersection Navigation","date":"2023-09-30","arxiv_id":"2310.08595","repositories_listed":0,"syntology":null},{"url":null,"slug":"pairwise-proximal-policy-optimization","title":"Pairwise Proximal Policy Optimization: Harnessing Relative Feedback for LLM Alignment","date":"2023-09-30","arxiv_id":"2310.00212","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-prompt-rewriting-for-personalized","title":"Learning to Rewrite Prompts for Personalized Text Generation","date":"2023-09-29","arxiv_id":"2310.00152","repositories_listed":0,"syntology":null},{"url":null,"slug":"dream-decentralized-reinforcement-learning","title":"DREAM: Decentralized Reinforcement Learning for Exploration and Efficient Energy Management in Multi-Robot Systems","date":"2023-09-29","arxiv_id":"2309.17433","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-generating-explanations-for-reinforcement","title":"On Generating Explanations for Reinforcement Learning Policies: An Empirical Study","date":"2023-09-29","arxiv_id":"2309.16960","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-node-selection-in","title":"Reinforcement Learning for Node Selection in Branch-and-Bound","date":"2023-09-29","arxiv_id":"2310.00112","repositories_listed":0,"syntology":null},{"url":null,"slug":"reliability-quantification-of-deep","title":"Reliability Quantification of Deep Reinforcement Learning-based Control","date":"2023-09-29","arxiv_id":"2309.16977","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsic-language-guided-exploration-for","title":"Intrinsic Language-Guided Exploration for Complex Long-Horizon Robotic Manipulation Tasks","date":"2023-09-28","arxiv_id":"2309.16347","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-offline-reinforcement-learning-certify","title":"Robust Offline Reinforcement Learning -- Certify the Confidence Interval","date":"2023-09-28","arxiv_id":"2309.16631","repositories_listed":0,"syntology":null},{"url":null,"slug":"age-minimization-in-massive-iot-via-uav-swarm","title":"Age Minimization in Massive IoT via UAV Swarm: A Multi-agent Reinforcement Learning Approach","date":"2023-09-26","arxiv_id":"2309.14757","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-trust-a-reinforcement-learning","title":"Decoding trust: A reinforcement learning perspective","date":"2023-09-26","arxiv_id":"2309.14598","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-latency-constrained-fronthaul","title":"Constrained Deep Reinforcement Learning for Fronthaul Compression Optimization","date":"2023-09-26","arxiv_id":"2309.15060","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-double-q-learning-for-continuous","title":"Adapting Double Q-Learning for Continuous Reinforcement Learning","date":"2023-09-25","arxiv_id":"2309.14471","repositories_listed":0,"syntology":null},{"url":null,"slug":"diser-designing-imaging-systems-with","title":"DISeR: Designing Imaging Systems with Reinforcement Learning","date":"2023-09-25","arxiv_id":"2309.13851","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-sensing-in-traffic-optimization","title":"Implicit Sensing in Traffic Optimization: Advanced Deep Reinforcement Learning Techniques","date":"2023-09-25","arxiv_id":"2309.14395","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-risk-aware-quadrupedal-locomotion","title":"Learning Risk-Aware Quadrupedal Locomotion using Distributional Reinforcement Learning","date":"2023-09-25","arxiv_id":"2309.14246","repositories_listed":0,"syntology":null},{"url":null,"slug":"ode-based-recurrent-model-free-reinforcement","title":"ODE-based Recurrent Model-free Reinforcement Learning for POMDPs","date":"2023-09-25","arxiv_id":"2309.14078","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-benefit-of-optimal-transport-for","title":"On the Benefit of Optimal Transport for Curriculum Reinforcement Learning","date":"2023-09-25","arxiv_id":"2309.14091","repositories_listed":0,"syntology":null},{"url":null,"slug":"tracking-control-for-a-spherical-pendulum-via","title":"Tracking Control for a Spherical Pendulum via Curriculum Reinforcement Learning","date":"2023-09-25","arxiv_id":"2309.14096","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-reachability-estimation-for-safe","title":"Iterative Reachability Estimation for Safe Reinforcement Learning","date":"2023-09-24","arxiv_id":"2309.13528","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-robust-header","title":"Reinforcement Learning for Robust Header Compression under Model Uncertainty","date":"2023-09-23","arxiv_id":"2309.13291","repositories_listed":0,"syntology":null},{"url":null,"slug":"diagnosing-and-exploiting-the-computational","title":"Diagnosing and exploiting the computational demands of videos games for deep reinforcement learning","date":"2023-09-22","arxiv_id":"2309.13181","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-function-design-for-crowd-simulation","title":"Reward Function Design for Crowd Simulation via Reinforcement Learning","date":"2023-09-22","arxiv_id":"2309.12841","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-reinforcement-learning-via","title":"Curriculum Reinforcement Learning via Morphology-Environment Co-Evolution","date":"2023-09-21","arxiv_id":"2309.12529","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-recover-for-safe-reinforcement","title":"Learning to Recover for Safe Reinforcement Learning","date":"2023-09-21","arxiv_id":"2309.11907","repositories_listed":0,"syntology":null},{"url":null,"slug":"rewiring-neurons-in-non-stationary","title":"Rewiring Neurons in Non-Stationary Environments","date":"2023-09-21","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-hierarchical-reinforcement-learning-for","title":"Safe Hierarchical Reinforcement Learning for CubeSat Task Scheduling Based on Energy Consumption","date":"2023-09-21","arxiv_id":"2309.12004","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-driven-patient-monitoring-with-multi-agent","title":"Adaptive Multi-Agent Deep Reinforcement Learning for Timely Healthcare Interventions","date":"2023-09-20","arxiv_id":"2309.10980","repositories_listed":0,"syntology":null},{"url":null,"slug":"delays-in-reinforcement-learning","title":"Delays in Reinforcement Learning","date":"2023-09-20","arxiv_id":"2309.11096","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-with-6","title":"Hierarchical reinforcement learning with natural language subgoals","date":"2023-09-20","arxiv_id":"2309.11564","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-quantum-architecture-search-1","title":"Differentiable Quantum Architecture Search for Quantum Reinforcement Learning","date":"2023-09-19","arxiv_id":"2309.10392","repositories_listed":0,"syntology":null},{"url":null,"slug":"framu-attention-based-machine-unlearning","title":"FRAMU: Attention-based Machine Unlearning using Federated Reinforcement Learning","date":"2023-09-19","arxiv_id":"2309.10283","repositories_listed":0,"syntology":null},{"url":null,"slug":"multicopy-reinforcement-learning-agents","title":"Multicopy Reinforcement Learning Agents","date":"2023-09-19","arxiv_id":"2309.10908","repositories_listed":0,"syntology":null},{"url":null,"slug":"pdrl-multi-agent-based-reinforcement-learning","title":"PDRL: Multi-Agent based Reinforcement Learning for Predictive Monitoring","date":"2023-09-19","arxiv_id":"2309.10576","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-liquidity-provision-in-uniswap-v3","title":"Adaptive Liquidity Provision in Uniswap V3 with Deep Reinforcement Learning","date":"2023-09-18","arxiv_id":"2309.10129","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-online-distillation-promoting-safe","title":"Guided Online Distillation: Promoting Safe Reinforcement Learning by Offline Demonstration","date":"2023-09-18","arxiv_id":"2309.09408","repositories_listed":0,"syntology":null},{"url":null,"slug":"mechanic-maker-2-0-reinforcement-learning-for","title":"Mechanic Maker 2.0: Reinforcement Learning for Evaluating Generated Rules","date":"2023-09-18","arxiv_id":"2309.09476","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-14","title":"Multi-Agent Deep Reinforcement Learning for Cooperative and Competitive Autonomous Vehicles using AutoDRIVE Ecosystem","date":"2023-09-18","arxiv_id":"2309.10007","repositories_listed":0,"syntology":null},{"url":null,"slug":"privileged-to-predicted-towards-sensorimotor","title":"Privileged to Predicted: Towards Sensorimotor Reinforcement Learning for Urban Driving","date":"2023-09-18","arxiv_id":"2309.09756","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-transformer-scalable-offline-reinforcement","title":"Q-Transformer: Scalable Offline Reinforcement Learning via Autoregressive Q-Functions","date":"2023-09-18","arxiv_id":"2309.10150","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-mildly-conservative-model-based","title":"DOMAIN: MilDly COnservative Model-BAsed OfflINe Reinforcement Learning","date":"2023-09-16","arxiv_id":"2309.08925","repositories_listed":0,"syntology":null},{"url":null,"slug":"gym-saturation-gymnasium-environments-for","title":"gym-saturation: Gymnasium environments for saturation provers (System description)","date":"2023-09-16","arxiv_id":"2309.09022","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-multi-agent-reinforcement-learning-for-3","title":"Deep Multi-Agent Reinforcement Learning for Decentralized Active Hypothesis Testing","date":"2023-09-14","arxiv_id":"2309.08477","repositories_listed":0,"syntology":null},{"url":null,"slug":"equivariant-data-augmentation-for","title":"Equivariant Data Augmentation for Generalization in Offline Reinforcement Learning","date":"2023-09-14","arxiv_id":"2309.07578","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-real-world-quadrupedal-locomotion-benchmark","title":"A Real-World Quadrupedal Locomotion Benchmark for Offline Reinforcement Learning","date":"2023-09-13","arxiv_id":"2309.16718","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-loss-adjusted-prioritized","title":"Attention Loss Adjusted Prioritized Experience Replay","date":"2023-09-13","arxiv_id":"2309.06684","repositories_listed":0,"syntology":null},{"url":null,"slug":"characterizing-speed-performance-of-multi","title":"Characterizing Speed Performance of Multi-Agent Reinforcement Learning","date":"2023-09-13","arxiv_id":"2309.07108","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-quantum-recurrent-reinforcement","title":"Efficient quantum recurrent reinforcement learning via quantum reservoir computing","date":"2023-09-13","arxiv_id":"2309.07339","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-dual","title":"Safe Reinforcement Learning with Dual Robustness","date":"2023-09-13","arxiv_id":"2309.06835","repositories_listed":0,"syntology":null},{"url":null,"slug":"fidelity-induced-interpretable-policy","title":"Fidelity-Induced Interpretable Policy Extraction for Reinforcement Learning","date":"2023-09-12","arxiv_id":"2309.06097","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-space-abstraction-in-hierarchical","title":"Goal Space Abstraction in Hierarchical Reinforcement Learning via Reachability Analysis","date":"2023-09-12","arxiv_id":"2309.07168","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-aware-reinforcement-learning-through","title":"Risk-Aware Reinforcement Learning through Optimal Transport Theory","date":"2023-09-12","arxiv_id":"2309.06239","repositories_listed":0,"syntology":null},{"url":null,"slug":"mind-the-uncertainty-risk-aware-and-actively","title":"Mind the Uncertainty: Risk-Aware and Actively Exploring Model-Based Reinforcement Learning","date":"2023-09-11","arxiv_id":"2309.05582","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-informed-reinforcement-learning-via","title":"Physics-informed reinforcement learning via probabilistic co-adjustment functions","date":"2023-09-11","arxiv_id":"2309.05404","repositories_listed":0,"syntology":null},{"url":null,"slug":"chasing-the-intruder-a-reinforcement-learning","title":"Chasing the Intruder: A Reinforcement Learning Approach for Tracking Intruder Drones","date":"2023-09-10","arxiv_id":"2309.05070","repositories_listed":0,"syntology":null}],"record_sha256":"1e729dff18aaddad06e2171fa89c32eda1f23d2fdfdba751960369d8546987ea","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}