{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/59","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":59,"pages_in_order":152,"rows_per_page":100,"rows":[5801,5900],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/58","next":"/task/reinforcement-learning-1/papers/60","papers":[{"url":null,"slug":"sampling-from-energy-based-policies-using","title":"Sampling from Energy-based Policies using Diffusion","date":"2024-10-02","arxiv_id":"2410.01312","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-reinforcement-learning-based-neural-1","title":"Scalable Reinforcement Learning-based Neural Architecture Search","date":"2024-10-02","arxiv_id":"2410.01431","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-autoencoders-reveal-temporal","title":"Sparse Autoencoders Reveal Temporal Difference Learning in Large Language Models","date":"2024-10-02","arxiv_id":"2410.01280","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalisation-via-dynamic-policy-fusion","title":"Personalisation via Dynamic Policy Fusion","date":"2024-09-30","arxiv_id":"2409.20016","repositories_listed":0,"syntology":null},{"url":"/paper/task-agnostic-pre-training-and-task-guided","slug":"task-agnostic-pre-training-and-task-guided","title":"Task-agnostic Pre-training and Task-guided Fine-tuning for Versatile Diffusion Planner","date":"2024-09-30","arxiv_id":"2409.19949","repositories_listed":0,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/task-agnostic-pre-training-and-task-guided#ran","syntology_url":"https://syntology.ai/paper/2409.19949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.19949"}},"official":null}},{"url":null,"slug":"upper-and-lower-bounds-for-distributionally","title":"Upper and Lower Bounds for Distributionally Robust Off-Dynamics Reinforcement Learning","date":"2024-09-30","arxiv_id":"2409.20521","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-on-riemann-hypothesis-with-cross","title":"Analysis on Riemann Hypothesis with Cross Entropy Optimization and Reasoning","date":"2024-09-29","arxiv_id":"2409.19790","repositories_listed":0,"syntology":null},{"url":null,"slug":"focus-on-what-matters-separated-models-for","title":"Focus On What Matters: Separated Models For Visual-Based RL Generalization","date":"2024-09-29","arxiv_id":"2410.10834","repositories_listed":0,"syntology":null},{"url":null,"slug":"grounded-curriculum-learning","title":"Grounded Curriculum Learning","date":"2024-09-29","arxiv_id":"2409.19816","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizing-consistency-policy-to-visual-rl","title":"Generalizing Consistency Policy to Visual RL with Prioritized Proximal Experience Regularization","date":"2024-09-28","arxiv_id":"2410.00051","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-bridge-the-gap-efficient-novelty","title":"Learning to Bridge the Gap: Efficient Novelty Recovery with Planning and Reinforcement Learning","date":"2024-09-28","arxiv_id":"2409.19226","repositories_listed":0,"syntology":null},{"url":null,"slug":"strongly-polynomial-time-and-validation","title":"Strongly-polynomial time and validation analysis of policy gradient methods","date":"2024-09-28","arxiv_id":"2409.19437","repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-aware-dynamic-cloud-workflow-scheduling","title":"Cost-Aware Dynamic Cloud Workflow Scheduling using Self-Attention and Evolutionary Reinforcement Learning","date":"2024-09-27","arxiv_id":"2409.18444","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-spectrum-efficiency-in-6g-satellite","title":"Enhancing Spectrum Efficiency in 6G Satellite Networks: A GAIL-Powered Policy Learning via Asynchronous Federated Inverse Reinforcement Learning","date":"2024-09-27","arxiv_id":"2409.18718","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporalpad-a-reinforcement-learning","title":"TemporalPaD: a reinforcement-learning framework for temporal feature representation and dimension reduction","date":"2024-09-27","arxiv_id":"2409.18597","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-multi-trait-essay-scoring-via","title":"Autoregressive Multi-trait Essay Scoring via Reinforcement Learning with Scoring-aware Multiple Rewards","date":"2024-09-26","arxiv_id":"2409.17472","repositories_listed":0,"syntology":null},{"url":null,"slug":"loopsr-looping-sim-and-real-for-lifelong","title":"LoopSR: Looping Sim-and-Real for Lifelong Policy Adaptation of Legged Robots","date":"2024-09-26","arxiv_id":"2409.17992","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-downlink-c-noma-transmission-with","title":"Optimizing Downlink C-NOMA Transmission with Movable Antennas: A DDPG-based Approach","date":"2024-09-26","arxiv_id":"2409.18281","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-random-measure-approach-to-reinforcement","title":"A random measure approach to reinforcement learning in continuous time","date":"2024-09-25","arxiv_id":"2409.17200","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-fractional-multi-agent-deep","title":"Asynchronous Fractional Multi-Agent Deep Reinforcement Learning for Age-Minimal Mobile Edge Computing","date":"2024-09-25","arxiv_id":"2409.16832","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-with-dynamics-autonomous-regulation","title":"Learning with Dynamics: Autonomous Regulation of UAV Based Communication Networks with Dynamic UAV Crew","date":"2024-09-25","arxiv_id":"2409.17139","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-and-distributional-reinforcement","title":"Offline and Distributional Reinforcement Learning for Radio Resource Management","date":"2024-09-25","arxiv_id":"2409.16764","repositories_listed":0,"syntology":null},{"url":null,"slug":"offripp-offline-rl-based-informative-path","title":"OffRIPP: Offline RL-based Informative Path Planning","date":"2024-09-25","arxiv_id":"2409.16830","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-orbit-servicing-for-spacecraft-collision","title":"On-orbit Servicing for Spacecraft Collision Avoidance With Autonomous Decision Making","date":"2024-09-25","arxiv_id":"2409.17125","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-space-mission-planning-a","title":"Revisiting Space Mission Planning: A Reinforcement Learning-Guided Approach for Multi-Debris Rendezvous","date":"2024-09-25","arxiv_id":"2409.16882","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-environments-and-language-with","title":"From Goal-Conditioned to Language-Conditioned Agents via Vision-Language Models","date":"2024-09-24","arxiv_id":"2409.16024","repositories_listed":0,"syntology":null},{"url":null,"slug":"development-and-validation-of-heparin-dosing","title":"Development and Validation of Heparin Dosing Policies Using an Offline Reinforcement Learning Algorithm","date":"2024-09-24","arxiv_id":"2409.15753","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-leaning-for-infinite","title":"Reinforcement Leaning for Infinite-Dimensional Systems","date":"2024-09-24","arxiv_id":"2409.15737","repositories_listed":0,"syntology":null},{"url":null,"slug":"whole-body-end-effector-pose-tracking","title":"Whole-body End-Effector Pose Tracking","date":"2024-09-24","arxiv_id":"2409.16048","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-agent-with-formal-goal-reaching","title":"A novel agent with formal goal-reaching guarantees: an experimental study with a mobile robot","date":"2024-09-23","arxiv_id":"2409.14867","repositories_listed":0,"syntology":null},{"url":null,"slug":"candere-coach-reinforcement-learning-from","title":"CANDERE-COACH: Reinforcement Learning from Noisy Feedback","date":"2024-09-23","arxiv_id":"2409.15521","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-saving-in-6g-o-ran-using-dqn-based","title":"Energy Saving in 6G O-RAN Using DQN-based xApp","date":"2024-09-23","arxiv_id":"2409.15098","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-routing-algorithm-over-sdn","title":"Intelligent Routing Algorithm over SDN: Reusable Reinforcement Learning Approach","date":"2024-09-23","arxiv_id":"2409.15226","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-enhanced-residual-policy-learning","title":"Physics Enhanced Residual Policy Learning (PERPL) for safety cruising in mixed traffic platooning under actuator and communication delay","date":"2024-09-23","arxiv_id":"2409.15595","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-distribution-aware-flow-matching-for","title":"A Distribution-Aware Flow-Matching for Generating Unstructured Data for Few-Shot Reinforcement Learning","date":"2024-09-21","arxiv_id":"2409.14178","repositories_listed":0,"syntology":null},{"url":null,"slug":"magics-adversarial-rl-with-minimax-actors","title":"MAGICS: Adversarial RL with Minimax Actors Guided by Implicit Critic Stackelberg for Convergent Neural Synthesis of Robot Safety","date":"2024-09-20","arxiv_id":"2409.13867","repositories_listed":0,"syntology":null},{"url":null,"slug":"omg-rl-offline-model-based-guided-reward","title":"OMG-RL:Offline Model-based Guided Reward Learning for Heparin Treatment","date":"2024-09-20","arxiv_id":"2409.13299","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-multi-agent-reinforcement-learning-5","title":"Scalable Multi-agent Reinforcement Learning for Factory-wide Dynamic Scheduling","date":"2024-09-20","arxiv_id":"2409.13571","repositories_listed":0,"syntology":null},{"url":null,"slug":"soloparkour-constrained-reinforcement","title":"SoloParkour: Constrained Reinforcement Learning for Visual Locomotion from Privileged Experience","date":"2024-09-20","arxiv_id":"2409.13678","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-the-zero-shot-capabilities-of-llms","title":"Assessing the Zero-Shot Capabilities of LLMs for Action Evaluation in RL","date":"2024-09-19","arxiv_id":"2409.12798","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-recognition-and-decision","title":"Disentangling Recognition and Decision Regrets in Image-Based Reinforcement Learning","date":"2024-09-19","arxiv_id":"2409.13108","repositories_listed":0,"syntology":null},{"url":null,"slug":"taco-rl-task-aware-prompt-compression","title":"TACO-RL: Task Aware Prompt Compression Optimization with Reinforcement Learning","date":"2024-09-19","arxiv_id":"2409.13035","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-central-role-of-the-loss-function-in","title":"The Central Role of the Loss Function in Reinforcement Learning","date":"2024-09-19","arxiv_id":"2409.12799","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-enhanced-state-reinforcement-learning","title":"An Enhanced-State Reinforcement Learning Algorithm for Multi-Task Fusion in Large-Scale Recommender Systems","date":"2024-09-18","arxiv_id":"2409.11678","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-quadratic-q-learning-using","title":"Data-Efficient Quadratic Q-Learning Using LMIs","date":"2024-09-18","arxiv_id":"2409.11986","repositories_listed":0,"syntology":null},{"url":null,"slug":"imrl-integrating-visual-physical-temporal-and","title":"IMRL: Integrating Visual, Physical, Temporal, and Geometric Representations for Enhanced Food Acquisition","date":"2024-09-18","arxiv_id":"2409.12092","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-as-an-improvement","title":"Reinforcement Learning as an Improvement Heuristic for Real-World Production Scheduling","date":"2024-09-18","arxiv_id":"2409.11933","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-environment-for-4","title":"A Reinforcement Learning Environment for Automatic Code Optimization in the MLIR Compiler","date":"2024-09-17","arxiv_id":"2409.11068","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-policy-actor-critic-reinforcement-learning","title":"On-policy Actor-Critic Reinforcement Learning for Multi-UAV Exploration","date":"2024-09-17","arxiv_id":"2409.11058","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-offline-adaptation-framework-for","title":"An Offline Adaptation Framework for Constrained Multi-Objective Reinforcement Learning","date":"2024-09-16","arxiv_id":"2409.09958","repositories_listed":0,"syntology":null},{"url":null,"slug":"instigating-cooperation-among-llm-agents","title":"Instigating Cooperation among LLM Agents Using Adaptive Information Modulation","date":"2024-09-16","arxiv_id":"2409.10372","repositories_listed":0,"syntology":null},{"url":null,"slug":"logic-synthesis-optimization-with-predictive","title":"Logic Synthesis Optimization with Predictive Self-Supervision via Causal Transformers","date":"2024-09-16","arxiv_id":"2409.10653","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-partial-observability-in-adaptive","title":"Mitigating Partial Observability in Adaptive Traffic Signal Control with Transformers","date":"2024-09-16","arxiv_id":"2409.10693","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-oriented-pruning-and-interpretation-of","title":"Safety-Oriented Pruning and Interpretation of Reinforcement Learning Policies","date":"2024-09-16","arxiv_id":"2409.10218","repositories_listed":0,"syntology":null},{"url":null,"slug":"kan-v-s-mlp-for-offline-reinforcement","title":"KAN v.s. MLP for Offline Reinforcement Learning","date":"2024-09-15","arxiv_id":"2409.09653","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-dimensionality-in-2d-rectangle","title":"Mitigating Dimensionality in 2D Rectangle Packing Problem under Reinforcement Learning Schema","date":"2024-09-15","arxiv_id":"2409.09677","repositories_listed":0,"syntology":null},{"url":null,"slug":"pip-loco-a-proprioceptive-infinite-horizon","title":"PIP-Loco: A Proprioceptive Infinite Horizon Planning Framework for Quadrupedal Robot Locomotion","date":"2024-09-14","arxiv_id":"2409.09441","repositories_listed":0,"syntology":null},{"url":null,"slug":"average-reward-maximum-entropy-reinforcement","title":"Average-Reward Maximum Entropy Reinforcement Learning for Underactuated Double Pendulum Tasks","date":"2024-09-13","arxiv_id":"2409.08938","repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-ensemble-for-variance-dependent-regret","title":"Batch Ensemble for Variance Dependent Regret in Stochastic Bandits","date":"2024-09-13","arxiv_id":"2409.08570","repositories_listed":0,"syntology":null},{"url":null,"slug":"cpl-critical-planning-step-learning-boosts","title":"CPL: Critical Plan Step Learning Boosts LLM Generalization in Reasoning Tasks","date":"2024-09-13","arxiv_id":"2409.08642","repositories_listed":0,"syntology":null},{"url":null,"slug":"quasimetric-value-functions-with-dense","title":"Quasimetric Value Functions with Dense Rewards","date":"2024-09-13","arxiv_id":"2409.08724","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-twin-for-autonomous-guided-vehicles","title":"Digital Twin for Autonomous Guided Vehicles based on Integrated Sensing and Communications","date":"2024-09-12","arxiv_id":"2409.08005","repositories_listed":0,"syntology":null},{"url":null,"slug":"hand-object-interaction-pretraining-from","title":"Hand-Object Interaction Pretraining from Videos","date":"2024-09-12","arxiv_id":"2409.08273","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-management-of-grid-interactive","title":"Optimal Management of Grid-Interactive Efficient Buildings via Safe Reinforcement Learning","date":"2024-09-12","arxiv_id":"2409.08132","repositories_listed":0,"syntology":null},{"url":null,"slug":"scores-as-actions-a-framework-of-fine-tuning","title":"Scores as Actions: a framework of fine-tuning diffusion models by continuous-time reinforcement learning","date":"2024-09-12","arxiv_id":"2409.08400","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-efficient-recursive-numeral-systems","title":"Learning Efficient Recursive Numeral Systems via Reinforcement Learning","date":"2024-09-11","arxiv_id":"2409.07170","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-decision-metamorphformer-a-casual","title":"Online Decision MetaMorphFormer: A Casual Transformer-Based Reinforcement Learning Framework of Universal Embodied Intelligence","date":"2024-09-11","arxiv_id":"2409.07341","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-augmentation-policies-from-a-model","title":"Automated Data Augmentation for Few-Shot Time Series Forecasting: A Reinforcement Learning Approach Guided by a Model Zoo","date":"2024-09-10","arxiv_id":"2409.06282","repositories_listed":0,"syntology":null},{"url":null,"slug":"superior-computer-chess-with-model-predictive","title":"Superior Computer Chess with Model Predictive Control, Reinforcement Learning, and Rollout","date":"2024-09-10","arxiv_id":"2409.06477","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-introduction-to-quantum-reinforcement","title":"An Introduction to Quantum Reinforcement Learning (QRL)","date":"2024-09-09","arxiv_id":"2409.05846","repositories_listed":0,"syntology":null},{"url":null,"slug":"bamdp-shaping-a-unified-theoretical-framework","title":"BAMDP Shaping: a Unified Theoretical Framework for Intrinsic Motivation and Reward Shaping","date":"2024-09-09","arxiv_id":"2409.05358","repositories_listed":0,"syntology":null},{"url":null,"slug":"betterbodies-reinforcement-learning-guided","title":"BetterBodies: Reinforcement Learning guided Diffusion for Antibody Sequence Design","date":"2024-09-09","arxiv_id":"2409.16298","repositories_listed":0,"syntology":null},{"url":null,"slug":"forward-kl-regularized-preference","title":"Forward KL Regularized Preference Optimization for Aligning Diffusion Policies","date":"2024-09-09","arxiv_id":"2409.05622","repositories_listed":0,"syntology":null},{"url":null,"slug":"markov-chain-variance-estimation-a-stochastic","title":"Markov Chain Variance Estimation: A Stochastic Approximation Approach","date":"2024-09-09","arxiv_id":"2409.05733","repositories_listed":0,"syntology":null},{"url":null,"slug":"causality-driven-reinforcement-learning-for","title":"Causality-Driven Reinforcement Learning for Joint Communication and Sensing","date":"2024-09-07","arxiv_id":"2409.15329","repositories_listed":0,"syntology":null},{"url":null,"slug":"lmgt-optimizing-exploration-exploitation","title":"Reward Guidance for Reinforcement Learning Tasks Based on Large Language Models: The LMGT Framework","date":"2024-09-07","arxiv_id":"2409.04744","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-adaptive-load","title":"Reinforcement Learning-Based Adaptive Load Balancing for Dynamic Cloud Environments","date":"2024-09-07","arxiv_id":"2409.04896","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-rate-maximization","title":"Reinforcement Learning for Rate Maximization in IRS-aided OWC Networks","date":"2024-09-07","arxiv_id":"2409.04842","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-directed-score-based-diffusion-models","title":"Reward-Directed Score-Based Diffusion Models via q-Learning","date":"2024-09-07","arxiv_id":"2409.04832","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-and-oracle-efficient-reinforcement","title":"Sample and Oracle Efficient Reinforcement Learning for MDPs with Linearly-Realizable Value Functions","date":"2024-09-07","arxiv_id":"2409.04840","repositories_listed":0,"syntology":null},{"url":null,"slug":"gaussian-mixture-model-q-functions-for","title":"Gaussian-Mixture-Model Q-Functions for Reinforcement Learning by Riemannian Optimization","date":"2024-09-06","arxiv_id":"2409.04374","repositories_listed":0,"syntology":null},{"url":null,"slug":"chirps-change-induced-regret-proxy-metrics","title":"CHIRPs: Change-Induced Regret Proxy metrics for Lifelong Reinforcement Learning","date":"2024-09-05","arxiv_id":"2409.03577","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-discrete-event-simulation-for","title":"Differentiable Discrete Event Simulation for Queuing Network Control","date":"2024-09-05","arxiv_id":"2409.03740","repositories_listed":0,"syntology":null},{"url":null,"slug":"infralib-enabling-reinforcement-learning-and","title":"InfraLib: Enabling Reinforcement Learning and Decision-Making for Large-Scale Infrastructure Management","date":"2024-09-05","arxiv_id":"2409.03167","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approach-to-optimizing","title":"Reinforcement Learning Approach to Optimizing Profilometric Sensor Trajectories for Surface Inspection","date":"2024-09-05","arxiv_id":"2409.03429","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-synchronization-and-policy-adaptation","title":"Robust synchronization and policy adaptation for networked heterogeneous agents","date":"2024-09-05","arxiv_id":"2409.03273","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-information-freshness-an-aoi","title":"Enhancing Information Freshness: An AoI Optimized Markov Decision Process Dedicated In the Underwater Task","date":"2024-09-04","arxiv_id":"2409.02424","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-as-efficient-reward","title":"Large Language Models as Efficient Reward Function Searchers for Custom-Environment Multi-Objective Reinforcement Learning","date":"2024-09-04","arxiv_id":"2409.02428","repositories_listed":0,"syntology":null},{"url":null,"slug":"tractable-offline-learning-of-regular","title":"Tractable Offline Learning of Regular Decision Processes","date":"2024-09-04","arxiv_id":"2409.02747","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-enabled-satellite","title":"Reinforcement Learning-enabled Satellite Constellation Reconfiguration and Retasking for Mission-Critical Applications","date":"2024-09-03","arxiv_id":"2409.02270","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-and-action-factorization-in-power-grids","title":"State and Action Factorization in Power Grids","date":"2024-09-03","arxiv_id":"2409.04467","repositories_listed":0,"syntology":null},{"url":null,"slug":"grounding-language-models-in-autonomous-loco","title":"Grounding Language Models in Autonomous Loco-manipulation Tasks","date":"2024-09-02","arxiv_id":"2409.01326","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundations-of-multivariate-distributional","title":"Foundations of Multivariate Distributional Reinforcement Learning","date":"2024-08-31","arxiv_id":"2409.00328","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-off-policy-reinforcement-learning-via","title":"Robust off-policy Reinforcement Learning via Soft Constrained Adversary","date":"2024-08-31","arxiv_id":"2409.00418","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovery-of-false-data-injection-schemes-on","title":"Discovery of False Data Injection Schemes on Frequency Controllers with Reinforcement Learning","date":"2024-08-30","arxiv_id":"2408.16958","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapshare-an-rl-based-dynamic-spectrum","title":"AdapShare: An RL-Based Dynamic Spectrum Sharing Solution for O-RAN","date":"2024-08-29","arxiv_id":"2408.16842","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-convergence-of-average-reward-q-learning","title":"On Convergence of Average-Reward Q-Learning in Weakly Communicating Markov Decision Processes","date":"2024-08-29","arxiv_id":"2408.16262","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-traffic-signal-control-using","title":"Reinforcement Learning for Adaptive Traffic Signal Control: Turn-Based and Time-Based Approaches to Reduce Congestion","date":"2024-08-28","arxiv_id":"2408.15751","repositories_listed":0,"syntology":null},{"url":null,"slug":"atari-gpt-investigating-the-capabilities-of","title":"Atari-GPT: Benchmarking Multimodal Large Language Models as Low-Level Policies in Atari Games","date":"2024-08-28","arxiv_id":"2408.15950","repositories_listed":0,"syntology":null},{"url":null,"slug":"skills-regularized-task-decomposition-for","title":"Skills Regularized Task Decomposition for Multi-task Offline Reinforcement Learning","date":"2024-08-28","arxiv_id":"2408.15593","repositories_listed":0,"syntology":null}],"record_sha256":"598f2ab935d5ec1e253f03e9ea760a02a0eecd5acd6234dbf3e11ca0bba5130f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}