{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/58","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":58,"pages_in_order":135,"rows_per_page":100,"rows":[5701,5800],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/57","next":"/task/reinforcement-learning-2/papers/59","papers":[{"url":null,"slug":"a-multi-step-loss-function-for-robust","title":"A Multi-step Loss Function for Robust Learning of the Dynamics in Model-based Reinforcement Learning","date":"2024-02-05","arxiv_id":"2402.03146","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for-dynamic","title":"A Reinforcement Learning Approach for Dynamic Rebalancing in Bike-Sharing System","date":"2024-02-05","arxiv_id":"2402.03589","repositories_listed":0,"syntology":null},{"url":null,"slug":"abstracted-trajectory-visualization-for","title":"Abstracted Trajectory Visualization for Explainability in Reinforcement Learning","date":"2024-02-05","arxiv_id":"2402.07928","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-the-impact-of-distribution-shift-on","title":"Assessing the Impact of Distribution Shift on Reinforcement Learning Performance","date":"2024-02-05","arxiv_id":"2402.03590","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-reinforcement-learning-for-quantum","title":"Curriculum reinforcement learning for quantum architecture search under hardware errors","date":"2024-02-05","arxiv_id":"2402.03500","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-autoregressive-density-nets-vs-neural","title":"Deep autoregressive density nets vs neural ensembles for model-based offline reinforcement learning","date":"2024-02-05","arxiv_id":"2402.02858","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-picker","title":"Deep Reinforcement Learning for Picker Routing Problem in Warehousing","date":"2024-02-05","arxiv_id":"2402.03525","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-abstract-visuomotor-mappings","title":"Learning to Abstract Visuomotor Mappings using Meta-Reinforcement Learning","date":"2024-02-05","arxiv_id":"2402.03072","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-21","title":"Multi-Agent Reinforcement Learning for Offloading Cellular Communications with Cooperating UAVs","date":"2024-02-05","arxiv_id":"2402.02957","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-actor-critic-learning-to","title":"Deep Exploration with PAC-Bayes","date":"2024-02-05","arxiv_id":"2402.03055","repositories_listed":0,"syntology":null},{"url":null,"slug":"utility-based-reinforcement-learning-unifying","title":"Utility-Based Reinforcement Learning: Unifying Single-objective and Multi-objective Reinforcement Learning","date":"2024-02-05","arxiv_id":"2402.02665","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-language-models-provide-promptable","title":"Vision-Language Models Provide Promptable Representations for Reinforcement Learning","date":"2024-02-05","arxiv_id":"2402.02651","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-inverse-reinforcement-learning","title":"Accelerating Inverse Reinforcement Learning with Expert Bootstrapping","date":"2024-02-04","arxiv_id":"2402.02608","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffstitch-boosting-offline-reinforcement","title":"DiffStitch: Boosting Offline Reinforcement Learning with Diffusion-based Trajectory Stitching","date":"2024-02-04","arxiv_id":"2402.02439","repositories_listed":0,"syntology":null},{"url":null,"slug":"evading-deep-learning-based-malware-detectors","title":"Evading Deep Learning-Based Malware Detectors via Obfuscation: A Deep Reinforcement Learning Approach","date":"2024-02-04","arxiv_id":"2402.02600","repositories_listed":0,"syntology":null},{"url":null,"slug":"interference-aware-emergent-random-access","title":"Interference-Aware Emergent Random Access Protocol for Downlink LEO Satellite Networks","date":"2024-02-04","arxiv_id":"2402.02350","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-virtues-of-pessimism-in-inverse","title":"The Virtues of Pessimism in Inverse Reinforcement Learning","date":"2024-02-04","arxiv_id":"2402.02616","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-constraint-formulations-in-safe","title":"A Survey of Constraint Formulations in Safe Reinforcement Learning","date":"2024-02-03","arxiv_id":"2402.02025","repositories_listed":0,"syntology":null},{"url":null,"slug":"brain-like-replay-naturally-emerges-in","title":"Brain-Like Replay Naturally Emerges in Reinforcement Learning Agents","date":"2024-02-02","arxiv_id":"2402.01467","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-for-routing","title":"Efficient Reinforcement Learning for Routing Jobs in Heterogeneous Queueing Systems","date":"2024-02-02","arxiv_id":"2402.01147","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-reinforcement-learning-with-self-1","title":"Near-Optimal Reinforcement Learning with Self-Play under Adaptivity Constraints","date":"2024-02-02","arxiv_id":"2402.01111","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-primal-dual-method-for-safe","title":"Adaptive Primal-Dual Method for Safe Reinforcement Learning","date":"2024-02-01","arxiv_id":"2402.00355","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-in-stochastic-environment-with-delays","title":"Control in Stochastic Environment with Delays: A Model-based Reinforcement Learning Approach","date":"2024-02-01","arxiv_id":"2402.00313","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-theoretic-techniques-for-online","title":"Control-Theoretic Techniques for Online Adaptation of Deep Neural Networks in Dynamical Systems","date":"2024-02-01","arxiv_id":"2402.00761","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-robot-sketching-an-application-of-deep-q","title":"Deep Robot Sketching: An application of Deep Q-Learning Networks for human-like sketching","date":"2024-02-01","arxiv_id":"2402.00676","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-conditional-diffusion-models-for","title":"Augmenting Offline Reinforcement Learning with State-only Interactions","date":"2024-02-01","arxiv_id":"2402.00807","repositories_listed":0,"syntology":null},{"url":null,"slug":"fm3q-factorized-multi-agent-minimax-q","title":"FM3Q: Factorized Multi-Agent MiniMax Q-Learning for Two-Team Zero-Sum Markov Game","date":"2024-02-01","arxiv_id":"2402.00738","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-policy-style-transfer","title":"Neural Policy Style Transfer","date":"2024-02-01","arxiv_id":"2402.00677","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-controller-to","title":"A Reinforcement Learning Based Controller to Minimize Forces on the Crutches of a Lower-Limb Exoskeleton","date":"2024-01-31","arxiv_id":"2402.00135","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-graph-for-multi-robot-social","title":"Attention Graph for Multi-Robot Social Navigation with Deep Reinforcement Learning","date":"2024-01-31","arxiv_id":"2401.17914","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-coordinated-concurrent-reinforcement","title":"Causal Coordinated Concurrent Reinforcement Learning","date":"2024-01-31","arxiv_id":"2401.18012","repositories_listed":0,"syntology":null},{"url":null,"slug":"circuit-partitioning-for-multi-core-quantum","title":"Circuit Partitioning for Multi-Core Quantum Architectures with Deep Reinforcement Learning","date":"2024-01-31","arxiv_id":"2401.17976","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-covert-routing-in-heterogeneous","title":"Decentralized Covert Routing in Heterogeneous Networks Using Reinforcement Learning","date":"2024-01-31","arxiv_id":"2402.10087","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-end-to-end-multi-task-dialogue","title":"Enhancing End-to-End Multi-Task Dialogue Systems: A Study on Intrinsic Motivation Reinforcement Learning Algorithms for Improved Training and Adaptability","date":"2024-01-31","arxiv_id":"2401.18040","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-attention-based-reinforcement-learning","title":"Graph Attention-based Reinforcement Learning for Trajectory Design and Resource Assignment in Multi-UAV Assisted Communication","date":"2024-01-31","arxiv_id":"2401.17880","repositories_listed":0,"syntology":null},{"url":null,"slug":"scheduled-curiosity-deep-dyna-q-efficient","title":"Scheduled Curiosity-Deep Dyna-Q: Efficient Exploration for Dialog Policy Learning","date":"2024-01-31","arxiv_id":"2402.00085","repositories_listed":0,"syntology":null},{"url":null,"slug":"extrinsicaly-rewarded-soft-q-imitation","title":"Extrinsicaly Rewarded Soft Q Imitation Learning with Discriminator","date":"2024-01-30","arxiv_id":"2401.16772","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-reinforcement-learning-from-human","title":"Improving Reinforcement Learning from Human Feedback with Efficient Reward Model Ensemble","date":"2024-01-30","arxiv_id":"2401.16635","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-versatile-dynamic","title":"Reinforcement Learning for Versatile, Dynamic, and Robust Bipedal Locomotion Control","date":"2024-01-30","arxiv_id":"2401.16889","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-reinforcement-learning-for-1","title":"Attention-based Reinforcement Learning for Combinatorial Optimization: Application to Job Shop Scheduling Problem","date":"2024-01-29","arxiv_id":"2401.16580","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergence-of-cooperation-under-punishment-a","title":"Emergence of cooperation under punishment: A reinforcement learning perspective","date":"2024-01-29","arxiv_id":"2401.16073","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-control-of-renewable-energy","title":"Optimal Control of Renewable Energy Communities subject to Network Peak Fees with Model Predictive Control and Reinforcement Learning Algorithms","date":"2024-01-29","arxiv_id":"2401.16321","repositories_listed":0,"syntology":null},{"url":null,"slug":"prepare-non-classical-collective-spin-state","title":"A Strategy for Preparing Quantum Squeezed States Using Reinforcement Learning","date":"2024-01-29","arxiv_id":"2401.16320","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-reinforcement-learning-for-linear","title":"Scalable Reinforcement Learning for Linear-Quadratic Control of Networks","date":"2024-01-29","arxiv_id":"2401.16183","repositories_listed":0,"syntology":null},{"url":null,"slug":"serl-a-software-suite-for-sample-efficient","title":"SERL: A Software Suite for Sample-Efficient Robotic Reinforcement Learning","date":"2024-01-29","arxiv_id":"2401.16013","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-analysis-of-on-policy","title":"Finite-Time Analysis of On-Policy Heterogeneous Federated Reinforcement Learning","date":"2024-01-27","arxiv_id":"2401.15273","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-trust-your-feelings-leveraging","title":"Learning to Trust Your Feelings: Leveraging Self-awareness in LLMs for Hallucination Mitigation","date":"2024-01-27","arxiv_id":"2401.15449","repositories_listed":0,"syntology":null},{"url":null,"slug":"social-interpretable-reinforcement-learning","title":"Social Interpretable Reinforcement Learning","date":"2024-01-27","arxiv_id":"2401.15480","repositories_listed":0,"syntology":null},{"url":null,"slug":"hi-core-hierarchical-knowledge-transfer-for","title":"Hierarchical Continual Reinforcement Learning via Large Language Model","date":"2024-01-25","arxiv_id":"2401.15098","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-and-optimization-of-epidemiological","title":"Modeling and Optimization of Epidemiological Control Policies Through Reinforcement Learning","date":"2024-01-25","arxiv_id":"2402.06640","repositories_listed":0,"syntology":null},{"url":null,"slug":"networked-multiagent-reinforcement-learning","title":"Peer-to-Peer Energy Trading of Solar and Energy Storage: A Networked Multiagent Reinforcement Learning Approach","date":"2024-01-25","arxiv_id":"2401.13947","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-by","title":"Sample Efficient Reinforcement Learning by Automatically Learning to Compose Subtasks","date":"2024-01-25","arxiv_id":"2401.14226","repositories_listed":0,"syntology":null},{"url":null,"slug":"scilab-rl-a-software-framework-for-efficient","title":"Scilab-RL: A software framework for efficient reinforcement learning and cognitive modeling research","date":"2024-01-25","arxiv_id":"2401.14488","repositories_listed":0,"syntology":null},{"url":null,"slug":"symbolic-equation-solving-via-reinforcement","title":"Symbolic Equation Solving via Reinforcement Learning","date":"2024-01-24","arxiv_id":"2401.13447","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-policy-iteration-algorithm-for","title":"A Novel Policy Iteration Algorithm for Nonlinear Continuous-Time H$\\infty$ Control Problem","date":"2024-01-23","arxiv_id":"2401.13014","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-safe-reinforcement-learning-algorithm-for","title":"A Safe Reinforcement Learning Algorithm for Supervisory Control of Power Plants","date":"2024-01-23","arxiv_id":"2401.13020","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-minimal-and-reusable-causal-state","title":"Building Minimal and Reusable Causal State Abstractions for Reinforcement Learning","date":"2024-01-23","arxiv_id":"2401.12497","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-communication-protocol-learning-for","title":"Emergent Communication Protocol Learning for Task Offloading in Industrial Internet of Things","date":"2024-01-23","arxiv_id":"2401.12914","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-d-policy-iteration-based-on-damped","title":"Model-Free $δ$-Policy Iteration Based on Damped Newton Method for Nonlinear Continuous-Time H$\\infty$ Tracking Control","date":"2024-01-23","arxiv_id":"2401.12882","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-relevance-filtered-linear-offline","title":"Reward-Relevance-Filtered Linear Offline Reinforcement Learning","date":"2024-01-23","arxiv_id":"2401.12934","repositories_listed":0,"syntology":null},{"url":null,"slug":"homerobot-open-vocabulary-mobile-manipulation-1","title":"HomeRobot Open Vocabulary Mobile Manipulation Challenge 2023 Participant Report (Team KuzHum)","date":"2024-01-22","arxiv_id":"2401.12048","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-covariate-shift-in-misspecified","title":"Mitigating Covariate Shift in Misspecified Regression with Applications to Reinforcement Learning","date":"2024-01-22","arxiv_id":"2401.12216","repositories_listed":0,"syntology":null},{"url":null,"slug":"p2dt-mitigating-forgetting-in-task","title":"P2DT: Mitigating Forgetting in task-incremental Learning with progressive prompt Decision Transformer","date":"2024-01-22","arxiv_id":"2401.11666","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-guided-reinforcement-learning-for","title":"Retrieval-Guided Reinforcement Learning for Boolean Circuit Minimization","date":"2024-01-22","arxiv_id":"2401.12205","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-and-generalized-end-to-end-autonomous","title":"Efficient and Generalized end-to-end Autonomous Driving System with Latent Deep Reinforcement Learning and Demonstrations","date":"2024-01-22","arxiv_id":"2401.11792","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-for-5","title":"Constrained Reinforcement Learning for Adaptive Controller Synchronization in Distributed SDN","date":"2024-01-21","arxiv_id":"2403.08775","repositories_listed":0,"syntology":null},{"url":null,"slug":"moma-model-based-mirror-ascent-for-offline","title":"MoMA: Model-based Mirror Ascent for Offline Reinforcement Learning","date":"2024-01-21","arxiv_id":"2401.11380","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-reinforcement-learning-for","title":"Large-scale Reinforcement Learning for Diffusion Models","date":"2024-01-20","arxiv_id":"2401.12244","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-empowered","title":"Deep Reinforcement Learning Empowered Activity-Aware Dynamic Health Monitoring Systems","date":"2024-01-19","arxiv_id":"2401.10794","repositories_listed":0,"syntology":null},{"url":null,"slug":"episodic-reinforcement-learning-with-expanded","title":"Episodic Reinforcement Learning with Expanded State-reward Space","date":"2024-01-19","arxiv_id":"2401.10516","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-dynamic-power-dispatch-with-high","title":"Stochastic Dynamic Power Dispatch with High Generalization and Few-Shot Adaption via Contextual Meta Graph Reinforcement Learning","date":"2024-01-19","arxiv_id":"2401.12235","repositories_listed":0,"syntology":null},{"url":null,"slug":"harnessing-density-ratios-for-online","title":"Harnessing Density Ratios for Online Reinforcement Learning","date":"2024-01-18","arxiv_id":"2401.09681","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-20","title":"Multi-Agent Reinforcement Learning for Maritime Operational Technology Cyber Security","date":"2024-01-18","arxiv_id":"2401.10149","repositories_listed":0,"syntology":null},{"url":null,"slug":"cascading-reinforcement-learning","title":"Cascading Reinforcement Learning","date":"2024-01-17","arxiv_id":"2401.08961","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-time-continuous-space-homeostatic","title":"Continuous Time Continuous Space Homeostatic Reinforcement Learning (CTCS-HRRL) : Towards Biological Self-Autonomous Agent","date":"2024-01-17","arxiv_id":"2401.08999","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-off-policy-reinforcement-learning-for","title":"Towards Off-Policy Reinforcement Learning for Ranking Policies with Human Feedback","date":"2024-01-17","arxiv_id":"2401.08959","repositories_listed":0,"syntology":null},{"url":null,"slug":"cnn-drl-with-shuffled-features-in-finance","title":"CNN-DRL with Shuffled Features in Finance","date":"2024-01-16","arxiv_id":"2402.03338","repositories_listed":0,"syntology":null},{"url":null,"slug":"cppo-continual-learning-for-reinforcement","title":"CPPO: Continual Learning for Reinforcement Learning with Human Feedback","date":"2024-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-quantum-natural-policy-gradients","title":"On Quantum Natural Policy Gradients","date":"2024-01-16","arxiv_id":"2401.08307","repositories_listed":0,"syntology":null},{"url":null,"slug":"prewrite-prompt-rewriting-with-reinforcement","title":"PRewrite: Prompt Rewriting with Reinforcement Learning","date":"2024-01-16","arxiv_id":"2401.08189","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-conversational","title":"Reinforcement Learning for Conversational Question Answering over Knowledge Graph","date":"2024-01-16","arxiv_id":"2401.08460","repositories_listed":0,"syntology":null},{"url":null,"slug":"revalued-regularised-ensemble-value","title":"REValueD: Regularised Ensemble Value-Decomposition for Factorisable Markov Decision Processes","date":"2024-01-16","arxiv_id":"2401.08850","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-continual-offline-reinforcement","title":"Solving Continual Offline Reinforcement Learning with Decision Transformer","date":"2024-01-16","arxiv_id":"2401.08478","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-multi-objective-optimization-with","title":"Constrained Multi-objective Optimization with Deep Reinforcement Learning Assisted Operator Selection","date":"2024-01-15","arxiv_id":"2402.12381","repositories_listed":0,"syntology":null},{"url":null,"slug":"go-explore-for-residential-energy-management","title":"Go-Explore for Residential Energy Management","date":"2024-01-15","arxiv_id":"2401.07710","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-ode-method-for-stochastic-approximation","title":"The ODE Method for Stochastic Approximation and Reinforcement Learning with Markovian Noise","date":"2024-01-15","arxiv_id":"2401.07844","repositories_listed":0,"syntology":null},{"url":null,"slug":"bet-explaining-deep-reinforcement-learning","title":"BET: Explaining Deep Reinforcement Learning through The Error-Prone Decisions","date":"2024-01-14","arxiv_id":"2401.07263","repositories_listed":0,"syntology":null},{"url":null,"slug":"drlc-reinforcement-learning-with-dense","title":"Beyond Sparse Rewards: Enhancing Reinforcement Learning with Language Model Critique in Text Generation","date":"2024-01-14","arxiv_id":"2401.07382","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-from-llm-feedback-to","title":"Reinforcement Learning from LLM Feedback to Counteract Goal Misgeneralization","date":"2024-01-14","arxiv_id":"2401.07181","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-environment-for-3","title":"A Reinforcement Learning Environment for Directed Quantum Circuit Synthesis","date":"2024-01-13","arxiv_id":"2401.07054","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-security-vulnerability-repair-using","title":"Code Security Vulnerability Repair Using Reinforcement Learning with Large Language Models","date":"2024-01-13","arxiv_id":"2401.07031","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-command-and-control-channels","title":"Discovering Command and Control Channels Using Reinforcement Learning","date":"2024-01-13","arxiv_id":"2401.07154","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-advantage-actor-critic-for","title":"Quantum Advantage Actor-Critic for Reinforcement Learning","date":"2024-01-13","arxiv_id":"2401.07043","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-scalable-train","title":"Reinforcement Learning for Scalable Train Timetable Rescheduling with Graph Representation","date":"2024-01-13","arxiv_id":"2401.06952","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-policy-gradient-subspaces","title":"Identifying Policy Gradient Subspaces","date":"2024-01-12","arxiv_id":"2401.06604","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-causal-entropy-inverse-reinforcement","title":"Maximum Causal Entropy Inverse Reinforcement Learning for Mean-Field Games","date":"2024-01-12","arxiv_id":"2401.06566","repositories_listed":0,"syntology":null},{"url":null,"slug":"unex-rl-reinforcing-long-term-rewards-in","title":"UNEX-RL: Reinforcing Long-Term Rewards in Multi-Stage Recommender Systems with UNidirectional EXecution","date":"2024-01-12","arxiv_id":"2401.06470","repositories_listed":0,"syntology":null},{"url":null,"slug":"bounds-on-the-price-of-feedback-for-mistake","title":"Bounds on the price of feedback for mistake-bounded online learning","date":"2024-01-11","arxiv_id":"2401.05794","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-reinforcement-learning-for-4","title":"Model-Free Reinforcement Learning for Automated Fluid Administration in Critical Care","date":"2024-01-11","arxiv_id":"2401.06299","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-aware-deep-reinforcement-learning-for","title":"Spatial-Aware Deep Reinforcement Learning for the Traveling Officer Problem","date":"2024-01-11","arxiv_id":"2401.05969","repositories_listed":0,"syntology":null}],"record_sha256":"82a9fc839f5db58118d7b71d80219b7c6c2f481613978df3e704a6f4911cc64b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}