{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/52","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":52,"pages_in_order":132,"rows_per_page":100,"rows":[5101,5200],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/51","next":"/task/reinforcement-learning/papers/53","papers":[{"url":null,"slug":"inverse-q-token-level-reinforcement-learning","title":"Inverse-Q*: Token Level Reinforcement Learning for Aligning Large Language Models Without Preference Data","date":"2024-08-27","arxiv_id":"2408.14874","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-stateful-value-factorization-in-multi","title":"On Stateful Value Factorization in Multi-Agent Reinforcement Learning","date":"2024-08-27","arxiv_id":"2408.15381","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimization-solution-functions-as","title":"Optimization Solution Functions as Deterministic Policies for Offline Reinforcement Learning","date":"2024-08-27","arxiv_id":"2408.15368","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-to-online-reinforcement-learning","title":"Unsupervised-to-Online Reinforcement Learning","date":"2024-08-27","arxiv_id":"2408.14785","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-reinforcement-learning","title":"A Survey on Reinforcement Learning Applications in SLAM","date":"2024-08-26","arxiv_id":"2408.14518","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-learning-to-plan","title":"Bridging the gap between Learning-to-plan, Motion Primitives and Safe Reinforcement Learning","date":"2024-08-26","arxiv_id":"2408.14063","repositories_listed":0,"syntology":null},{"url":null,"slug":"equivariant-reinforcement-learning-under","title":"Equivariant Reinforcement Learning under Partial Observability","date":"2024-08-26","arxiv_id":"2408.14336","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-9","title":"Model-Based Reinforcement Learning for Control of Strongly-Disturbed Unsteady Aerodynamic Flows","date":"2024-08-26","arxiv_id":"2408.14685","repositories_listed":0,"syntology":null},{"url":null,"slug":"relexs-reinforcement-learning-explanations","title":"ReLExS: Reinforcement Learning Explanations for Stackelberg No-Regret Learners","date":"2024-08-26","arxiv_id":"2408.14086","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforced-dag-learning-without","title":"Reinforcement Learning for Causal Discovery without Acyclicity Constraints","date":"2024-08-24","arxiv_id":"2408.13448","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-state-disentanglement-in-causal","title":"Rethinking State Disentanglement in Causal Reinforcement Learning","date":"2024-08-24","arxiv_id":"2408.13498","repositories_listed":0,"syntology":null},{"url":null,"slug":"thresholded-lexicographic-ordered","title":"Thresholded Lexicographic Ordered Multiobjective Reinforcement Learning","date":"2024-08-24","arxiv_id":"2408.13493","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-based-episodes-augmentation-for","title":"Diffusion-based Episodes Augmentation for Offline Multi-Agent Reinforcement Learning","date":"2024-08-23","arxiv_id":"2408.13092","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-context-learning-with-reinforcement","title":"In-Context Learning with Reinforcement Learning for Incomplete Utterance Rewriting","date":"2024-08-23","arxiv_id":"2408.13028","repositories_listed":0,"syntology":null},{"url":null,"slug":"mastering-the-digital-art-of-war-developing","title":"Mastering the Digital Art of War: Developing Intelligent Combat Simulation Agents for Wargaming Using Hierarchical Reinforcement Learning","date":"2024-08-23","arxiv_id":"2408.13333","repositories_listed":0,"syntology":null},{"url":null,"slug":"reduce-reuse-recycle-categories-for","title":"Reduce, Reuse, Recycle: Categories for Compositional Reinforcement Learning","date":"2024-08-23","arxiv_id":"2408.13376","repositories_listed":0,"syntology":null},{"url":null,"slug":"sambo-rl-shifts-aware-model-based-offline","title":"SAMBO-RL: Shifts-aware Model-based Offline Reinforcement Learning","date":"2024-08-23","arxiv_id":"2408.12830","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-safe-and-efficient-self-evolving-algorithm","title":"A Safe and Efficient Self-evolving Algorithm for Decision-making and Control of Autonomous Driving Systems","date":"2024-08-22","arxiv_id":"2408.12187","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adaptation-for-offline-reinforcement","title":"Domain Adaptation for Offline Reinforcement Learning with Limited Samples","date":"2024-08-22","arxiv_id":"2408.12136","repositories_listed":0,"syntology":null},{"url":null,"slug":"pareto-inverse-reinforcement-learning-for","title":"Pareto Inverse Reinforcement Learning for Diverse Expert Policy Generation","date":"2024-08-22","arxiv_id":"2408.12110","repositories_listed":0,"syntology":null},{"url":null,"slug":"pcgrl-scaling-control-and-generalization-in","title":"PCGRL+: Scaling, Control and Generalization in Reinforcement Learning Level Generators","date":"2024-08-22","arxiv_id":"2408.12525","repositories_listed":0,"syntology":null},{"url":null,"slug":"advances-in-preference-based-reinforcement","title":"Advances in Preference-based Reinforcement Learning: A Review","date":"2024-08-21","arxiv_id":"2408.11943","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-exploration-and-discriminative","title":"Efficient Exploration and Discriminative World Model Learning with an Object-Centric Abstraction","date":"2024-08-21","arxiv_id":"2408.11816","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-part-based-representations-for","title":"Using Part-based Representations for Explainable Deep Reinforcement Learning","date":"2024-08-21","arxiv_id":"2408.11455","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-model-based-reinforcement-learning","title":"Offline Model-Based Reinforcement Learning with Anti-Exploration","date":"2024-08-20","arxiv_id":"2408.10713","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-evolution-of-reinforcement-learning-in","title":"The Evolution of Reinforcement Learning in Quantitative Finance: A Survey","date":"2024-08-20","arxiv_id":"2408.10932","repositories_listed":0,"syntology":null},{"url":null,"slug":"demystifying-reinforcement-learning-in","title":"Demystifying Reinforcement Learning in Production Scheduling via Explainable AI","date":"2024-08-19","arxiv_id":"2408.09841","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-exploration-in-deep-reinforcement","title":"Efficient Exploration in Deep Reinforcement Learning: A Novel Bayesian Actor-Critic Algorithm","date":"2024-08-19","arxiv_id":"2408.10055","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-in-1","title":"Efficient Reinforcement Learning in Probabilistic Reward Machines","date":"2024-08-19","arxiv_id":"2408.10381","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-reinforcement-learning-through","title":"Enhancing Reinforcement Learning Through Guided Search","date":"2024-08-19","arxiv_id":"2408.10113","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpt-augmented-reinforcement-learning-with","title":"GARLIC: GPT-Augmented Reinforcement Learning with Intelligent Control for Vehicle Dispatching","date":"2024-08-19","arxiv_id":"2408.10286","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalizing-reinforcement-learning-from","title":"Personalizing Reinforcement Learning from Human Feedback with Variational Preference Learning","date":"2024-08-19","arxiv_id":"2408.10075","repositories_listed":0,"syntology":null},{"url":null,"slug":"world-models-increase-autonomy-in","title":"World Models Increase Autonomy in Reinforcement Learning","date":"2024-08-19","arxiv_id":"2408.09807","repositories_listed":0,"syntology":null},{"url":null,"slug":"ancestral-reinforcement-learning-unifying","title":"Ancestral Reinforcement Learning: Unifying Zeroth-Order Optimization and Genetic Algorithms for Reinforcement Learning","date":"2024-08-18","arxiv_id":"2408.09493","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-local-views-global-state-inference","title":"Beyond Local Views: Global State Inference with Diffusion Models for Cooperative Multi-Agent Reinforcement Learning","date":"2024-08-18","arxiv_id":"2408.09501","repositories_listed":0,"syntology":null},{"url":null,"slug":"directed-exploration-in-reinforcement","title":"Directed Exploration in Reinforcement Learning from Linear Temporal Logic","date":"2024-08-18","arxiv_id":"2408.09495","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploratory-optimal-stopping-a-singular","title":"Exploratory Optimal Stopping: A Singular Control Formulation","date":"2024-08-18","arxiv_id":"2408.09335","repositories_listed":0,"syntology":null},{"url":null,"slug":"refine-lm-mitigating-language-model","title":"REFINE-LM: Mitigating Language Model Stereotypes via Reinforcement Learning","date":"2024-08-18","arxiv_id":"2408.09489","repositories_listed":0,"syntology":null},{"url":null,"slug":"qedcartographer-automating-formal","title":"QEDCartographer: Automating Formal Verification Using Reward-Free Reinforcement Learning","date":"2024-08-17","arxiv_id":"2408.09237","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-verifiably-robust-agents-using-set","title":"Training Verifiably Robust Agents Using Set-Based Reinforcement Learning","date":"2024-08-17","arxiv_id":"2408.09112","repositories_listed":0,"syntology":null},{"url":null,"slug":"cat-caution-aware-transfer-in-reinforcement","title":"CAT: Caution Aware Transfer in Reinforcement Learning via Distributional Risk","date":"2024-08-16","arxiv_id":"2408.08812","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-multi-intentional-inverse-reinforcement","title":"Deep multi-intentional inverse reinforcement learning for cognitive multi-function radar inverse cognition","date":"2024-08-16","arxiv_id":"2408.08478","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-multi-policy-evaluation-for","title":"Efficient Multi-Policy Evaluation for Reinforcement Learning","date":"2024-08-16","arxiv_id":"2408.08706","repositories_listed":0,"syntology":null},{"url":null,"slug":"d5rl-diverse-datasets-for-data-driven-deep","title":"D5RL: Diverse Datasets for Data-Driven Deep Reinforcement Learning","date":"2024-08-15","arxiv_id":"2408.08441","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifelong-reinforcement-learning-via","title":"Lifelong Reinforcement Learning via Neuromodulation","date":"2024-08-15","arxiv_id":"2408.08446","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-nested-graph-reinforcement-learning-based","title":"A Nested Graph Reinforcement Learning-based Decision-making Strategy for Eco-platooning","date":"2024-08-14","arxiv_id":"2408.07578","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-behavioral-ai-reinforcement-learning","title":"Adaptive Behavioral AI: Reinforcement Learning to Enhance Pharmacy Services","date":"2024-08-14","arxiv_id":"2408.07647","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-reinforcement-learning-with-high","title":"Off-Policy Reinforcement Learning with High Dimensional Reward","date":"2024-08-14","arxiv_id":"2408.07660","repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-deep-reinforcement-learning-for","title":"Residual Deep Reinforcement Learning for Inverter-based Volt-Var Control","date":"2024-08-13","arxiv_id":"2408.06790","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-in-context-reinforcement","title":"Retrieval-Augmented Hierarchical in-Context Reinforcement Learning and Hindsight Modular Reflections for Task Planning with LLMs","date":"2024-08-12","arxiv_id":"2408.06520","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-optimization-of-curriculum-learning","title":"Online Optimization of Curriculum Learning Schedules using Evolutionary Optimization","date":"2024-08-12","arxiv_id":"2408.06068","repositories_listed":0,"syntology":null},{"url":null,"slug":"curling-the-dream-contrastive-representations","title":"CURLing the Dream: Contrastive Representations for World Modeling in Reinforcement Learning","date":"2024-08-11","arxiv_id":"2408.05781","repositories_listed":0,"syntology":null},{"url":null,"slug":"root-cause-attribution-of-delivery-risks-via","title":"Root Cause Attribution of Delivery Risks via Causal Discovery with Reinforcement Learning","date":"2024-08-11","arxiv_id":"2408.05860","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-multi-agent-deep-reinforcement-4","title":"Cooperative Multi-Agent Deep Reinforcement Learning in Content Ranking Optimization","date":"2024-08-08","arxiv_id":"2408.04251","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-reinforcement-learning-breaks-sample","title":"Hybrid Reinforcement Learning Breaks Sample Size Barriers in Linear MDPs","date":"2024-08-08","arxiv_id":"2408.04526","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowpc-knowledge-driven-programmatic","title":"KnowPC: Knowledge-Driven Programmatic Reinforcement Learning for Zero-shot Coordination","date":"2024-08-08","arxiv_id":"2408.04336","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-driven-approach-for-sustainable-extraction","title":"AI-Driven approach for sustainable extraction of earth's subsurface renewable energy while minimizing seismic activity","date":"2024-08-07","arxiv_id":"2408.03664","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-03029","title":"Highly Efficient Self-Adaptive Reward Shaping for Reinforcement Learning","date":"2024-08-06","arxiv_id":"2408.03029","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-03084","title":"Research on Autonomous Driving Decision-making Strategies based Deep Reinforcement Learning","date":"2024-08-06","arxiv_id":"2408.03084","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-03088","title":"QADQN: Quantum Attention Deep Q-Network for Financial Market Prediction","date":"2024-08-06","arxiv_id":"2408.03088","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-03166","title":"CADRL: Category-aware Dual-agent Reinforcement Learning for Explainable Recommendations over Knowledge Graphs","date":"2024-08-06","arxiv_id":"2408.03166","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02349","title":"Active Sensing of Knee Osteoarthritis Progression with Reinforcement Learning","date":"2024-08-05","arxiv_id":"2408.02349","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02022","title":"Scenario-based Thermal Management Parametrization Through Deep Reinforcement Learning","date":"2024-08-04","arxiv_id":"2408.02022","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02165","title":"SelfBC: Self Behavior Cloning for Offline Reinforcement Learning","date":"2024-08-04","arxiv_id":"2408.02165","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01072","title":"A Survey on Self-play Methods in Reinforcement Learning","date":"2024-08-02","arxiv_id":"2408.01072","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01187","title":"Optimizing Variational Quantum Circuits Using Metaheuristic Strategies in Reinforcement Learning","date":"2024-08-02","arxiv_id":"2408.01187","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01188","title":"Multi-Objective Deep Reinforcement Learning for Optimisation in Autonomous Systems","date":"2024-08-02","arxiv_id":"2408.01188","repositories_listed":0,"syntology":null},{"url":null,"slug":"2407-21260","title":"Bellman Unbiasedness: Toward Provably Efficient Distributional Reinforcement Learning with General Value Function Approximation","date":"2024-07-31","arxiv_id":"2407.21260","repositories_listed":0,"syntology":null},{"url":null,"slug":"architectural-influence-on-variational","title":"Architectural Influence on Variational Quantum Circuits in Multi-Agent Reinforcement Learning: Evolutionary Strategies for Optimization","date":"2024-07-30","arxiv_id":"2407.20739","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-choose-a-reinforcement-learning","title":"How to Choose a Reinforcement-Learning Algorithm","date":"2024-07-30","arxiv_id":"2407.20917","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-generalizable-reinforcement-learning-1","title":"Towards Generalizable Reinforcement Learning via Causality-Guided Self-Adaptive Representations","date":"2024-07-30","arxiv_id":"2407.20651","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-method-for-fast-autonomy-transfer-in","title":"A Method for Fast Autonomy Transfer in Reinforcement Learning","date":"2024-07-29","arxiv_id":"2407.20466","repositories_listed":0,"syntology":null},{"url":null,"slug":"anomalous-state-sequence-modeling-to-enhance","title":"Anomalous State Sequence Modeling to Enhance Safety in Reinforcement Learning","date":"2024-07-29","arxiv_id":"2407.19860","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-dice-in-sample-diffusion-guidance","title":"Diffusion-DICE: In-Sample Diffusion Guidance for Offline Reinforcement Learning","date":"2024-07-29","arxiv_id":"2407.20109","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-machine-learning-architecture-search","title":"Quantum Machine Learning Architecture Search via Deep Reinforcement Learning","date":"2024-07-29","arxiv_id":"2407.20147","repositories_listed":0,"syntology":null},{"url":null,"slug":"empowering-clinicians-with-medical-decision","title":"Empowering Clinicians with Medical Decision Transformers: A Framework for Sepsis Treatment","date":"2024-07-28","arxiv_id":"2407.19380","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-interpretability-of-codebooks-in-model","title":"The Interpretability of Codebooks in Model-Based Reinforcement Learning is Limited","date":"2024-07-28","arxiv_id":"2407.19532","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-benefits-of-pixel-based-hierarchical","title":"On the benefits of pixel-based hierarchical policies for task generalization","date":"2024-07-27","arxiv_id":"2407.19142","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-sustainable-energy","title":"Reinforcement Learning for Sustainable Energy: A Survey","date":"2024-07-26","arxiv_id":"2407.18597","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-cross-environment-hyperparameter-setting","title":"The Cross-environment Hyperparameter Setting Benchmark for Reinforcement Learning","date":"2024-07-26","arxiv_id":"2407.18840","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-quantum-architecture-search-in","title":"Differentiable Quantum Architecture Search in Asynchronous Quantum Reinforcement Learning","date":"2024-07-25","arxiv_id":"2407.18202","repositories_listed":0,"syntology":null},{"url":null,"slug":"principal-agent-reinforcement-learning","title":"Principal-Agent Reinforcement Learning: Orchestrating AI Agents with Contracts","date":"2024-07-25","arxiv_id":"2407.18074","repositories_listed":0,"syntology":null},{"url":null,"slug":"movelight-enhancing-traffic-signal-control","title":"MoveLight: Enhancing Traffic Signal Control through Movement-Centric Deep Reinforcement Learning","date":"2024-07-24","arxiv_id":"2407.17303","repositories_listed":0,"syntology":null},{"url":null,"slug":"pretrained-visual-representations-in","title":"Pretrained Visual Representations in Reinforcement Learning","date":"2024-07-24","arxiv_id":"2407.17238","repositories_listed":0,"syntology":null},{"url":null,"slug":"sonic-safe-social-navigation-with-adaptive","title":"SoNIC: Safe Social Navigation with Adaptive Conformal Inference and Constrained Reinforcement Learning","date":"2024-07-24","arxiv_id":"2407.17460","repositories_listed":0,"syntology":null},{"url":null,"slug":"traversing-pareto-optimal-policies-provably","title":"Traversing Pareto Optimal Policies: Provably Efficient Multi-Objective Reinforcement Learning","date":"2024-07-24","arxiv_id":"2407.17466","repositories_listed":0,"syntology":null},{"url":null,"slug":"odgr-online-dynamic-goal-recognition","title":"ODGR: Online Dynamic Goal Recognition","date":"2024-07-23","arxiv_id":"2407.16220","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-overview-of-reward-engineering","title":"Comprehensive Overview of Reward Engineering and Shaping in Advancing Reinforcement Learning Applications","date":"2024-07-22","arxiv_id":"2408.10215","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-replay-memory-architectures-in","title":"Efficient Replay Memory Architectures in Multi-Agent Reinforcement Learning for Traffic Congestion Control","date":"2024-07-22","arxiv_id":"2407.16034","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-hardware-fault-tolerance-in","title":"Enhancing Hardware Fault Tolerance in Machines with Reinforcement Learning Policy Gradient Algorithms","date":"2024-07-21","arxiv_id":"2407.15283","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-deep-reinforcement-learning","title":"Mitigating Deep Reinforcement Learning Backdoors in the Neural Activation Space","date":"2024-07-21","arxiv_id":"2407.15168","repositories_listed":0,"syntology":null},{"url":null,"slug":"rocket-landing-control-with-random-annealing","title":"Rocket Landing Control with Random Annealing Jump Start Reinforcement Learning","date":"2024-07-21","arxiv_id":"2407.15083","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-abstraction-in-reinforcement-1","title":"Temporal Abstraction in Reinforcement Learning with Offline Data","date":"2024-07-21","arxiv_id":"2407.15241","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapt2reward-adapting-video-language-models","title":"Adapt2Reward: Adapting Video-Language Models to Generalizable Robotic Rewards via Failure Prompts","date":"2024-07-20","arxiv_id":"2407.14872","repositories_listed":0,"syntology":null},{"url":null,"slug":"phase-re-service-in-reinforcement-learning","title":"Phase Re-service in Reinforcement Learning Traffic Signal Control","date":"2024-07-20","arxiv_id":"2407.14775","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperparameter-optimization-for-driving","title":"Hyperparameter Optimization for Driving Strategies Based on Reinforcement Learning","date":"2024-07-19","arxiv_id":"2407.14262","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-policy-evaluation-algorithms-in","title":"On Policy Evaluation Algorithms in Distributional Reinforcement Learning","date":"2024-07-19","arxiv_id":"2407.14175","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01429","title":"An Agile Adaptation Method for Multi-mode Vehicle Communication Networks","date":"2024-07-18","arxiv_id":"2408.01429","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-strategy-to-automate","title":"A reinforcement learning strategy to automate and accelerate h/p-multigrid solvers","date":"2024-07-18","arxiv_id":"2407.15872","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-and-bridging-the-gap-between","title":"Analyzing and Bridging the Gap between Maximizing Total Reward and Discounted Reward in Deep Reinforcement Learning","date":"2024-07-18","arxiv_id":"2407.13279","repositories_listed":0,"syntology":null}],"record_sha256":"4684268132cd6a7838566a6377482b659a6f6538eb3ad0e8d5e981856928074a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}