{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/54","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":54,"pages_in_order":135,"rows_per_page":100,"rows":[5301,5400],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/53","next":"/task/reinforcement-learning-2/papers/55","papers":[{"url":null,"slug":"generative-ai-for-deep-reinforcement-learning","title":"Generative AI for Deep Reinforcement Learning: Framework, Analysis, and Use Cases","date":"2024-05-31","arxiv_id":"2405.20568","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-sociohydrology","title":"Reinforcement Learning for Sociohydrology","date":"2024-05-31","arxiv_id":"2405.20772","repositories_listed":0,"syntology":null},{"url":null,"slug":"bilevel-reinforcement-learning-via-the","title":"Bilevel reinforcement learning via the development of hyper-gradient without lower-level convexity","date":"2024-05-30","arxiv_id":"2405.19697","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-feature-selection-in-medical","title":"Dynamic feature selection in medical predictive monitoring by reinforcement learning","date":"2024-05-30","arxiv_id":"2405.19729","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-stimuli-generation-using","title":"Efficient Stimuli Generation using Reinforcement Learning in Design Verification","date":"2024-05-30","arxiv_id":"2405.19815","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-reinforcement-learning-framework-for","title":"Hybrid Reinforcement Learning Framework for Mixed-Variable Problems","date":"2024-05-30","arxiv_id":"2405.20500","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-random-demonstrations-offline","title":"Learning from Random Demonstrations: Offline Reinforcement Learning with Importance-Sampled Diffusion Models","date":"2024-05-30","arxiv_id":"2405.19878","repositories_listed":0,"syntology":null},{"url":null,"slug":"metacurl-non-stationary-concave-utility","title":"MetaCURL: Non-stationary Concave Utility Reinforcement Learning","date":"2024-05-30","arxiv_id":"2405.19807","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-as-a-monotone-scheme","title":"Q-learning as a monotone scheme","date":"2024-05-30","arxiv_id":"2405.20538","repositories_listed":0,"syntology":null},{"url":null,"slug":"randomized-exploration-for-reinforcement-1","title":"Randomized Exploration for Reinforcement Learning with Multinomial Logistic Function Approximation","date":"2024-05-30","arxiv_id":"2405.20165","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-multi-agent-reinforcement-learning-with-1","title":"Safe Multi-agent Reinforcement Learning with Natural Language Constraints","date":"2024-05-30","arxiv_id":"2405.20018","repositories_listed":0,"syntology":null},{"url":null,"slug":"sleepernets-universal-backdoor-poisoning","title":"SleeperNets: Universal Backdoor Poisoning Attacks Against Reinforcement Learning Agents","date":"2024-05-30","arxiv_id":"2405.20539","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-discretization-based-non-episodic","title":"Policy Zooming: Adaptive Discretization-based Infinite-Horizon Average-Reward Reinforcement Learning","date":"2024-05-29","arxiv_id":"2405.18793","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-household-robotics-deep-interactive","title":"Advancing Household Robotics: Deep Interactive Reinforcement Learning for Efficient Training and Enhanced Performance","date":"2024-05-29","arxiv_id":"2405.18687","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-preference-based-reinforcement-1","title":"Efficient Preference-based Reinforcement Learning via Aligned Experience Estimation","date":"2024-05-29","arxiv_id":"2405.18688","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-concave-utility-reinforcement","title":"Inverse Concave-Utility Reinforcement Learning is Inverse Game Theory","date":"2024-05-29","arxiv_id":"2405.19024","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-regularised-reinforcement-learning","title":"Offline Regularised Reinforcement Learning for Large Language Models Alignment","date":"2024-05-29","arxiv_id":"2405.19107","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-vehicular-networks-with","title":"Optimizing Vehicular Networks with Variational Quantum Circuits-based Reinforcement Learning","date":"2024-05-29","arxiv_id":"2405.18984","repositories_listed":0,"syntology":null},{"url":null,"slug":"preferred-action-optimized-diffusion-policies","title":"Preferred-Action-Optimized Diffusion Policies for Offline Reinforcement Learning","date":"2024-05-29","arxiv_id":"2405.18729","repositories_listed":0,"syntology":null},{"url":null,"slug":"rich-observation-reinforcement-learning-with","title":"Rich-Observation Reinforcement Learning with Continuous Latent Dynamics","date":"2024-05-29","arxiv_id":"2405.19269","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-risk-safe-reinforcement-learning","title":"Spectral-Risk Safe Reinforcement Learning with Convergence Guarantees","date":"2024-05-29","arxiv_id":"2405.18698","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-incentivized-preference-optimization-a","title":"Value-Incentivized Preference Optimization: A Unified Approach to Online and Offline RLHF","date":"2024-05-29","arxiv_id":"2405.19320","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-reinforcement-learning-in-energy-systems","title":"Why Reinforcement Learning in Energy Systems Needs Explanations","date":"2024-05-29","arxiv_id":"2405.18823","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-pontryagin-perspective-on-reinforcement","title":"A Pontryagin Perspective on Reinforcement Learning","date":"2024-05-28","arxiv_id":"2405.18100","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-horizon-actor-critic-for-policy","title":"Adaptive Horizon Actor-Critic for Policy Learning in Contact-Rich Differentiable Simulation","date":"2024-05-28","arxiv_id":"2405.17784","repositories_listed":0,"syntology":null},{"url":null,"slug":"highway-reinforcement-learning","title":"Highway Reinforcement Learning","date":"2024-05-28","arxiv_id":"2405.18289","repositories_listed":0,"syntology":null},{"url":null,"slug":"mutation-bias-learning-in-games","title":"Mutation-Bias Learning in Games","date":"2024-05-28","arxiv_id":"2405.18190","repositories_listed":0,"syntology":null},{"url":null,"slug":"biological-neurons-compete-with-deep","title":"Biological Neurons Compete with Deep Reinforcement Learning in Sample Efficiency in a Simulated Gameworld","date":"2024-05-27","arxiv_id":"2405.16946","repositories_listed":0,"syntology":null},{"url":null,"slug":"opinion-guided-reinforcement-learning","title":"Opinion-Guided Reinforcement Learning","date":"2024-05-27","arxiv_id":"2405.17287","repositories_listed":0,"syntology":null},{"url":null,"slug":"oracle-efficient-reinforcement-learning-for","title":"Oracle-Efficient Reinforcement Learning for Max Value Ensembles","date":"2024-05-27","arxiv_id":"2405.16739","repositories_listed":0,"syntology":null},{"url":null,"slug":"partial-models-for-building-adaptive-model","title":"Partial Models for Building Adaptive Model-Based Reinforcement Learning Agents","date":"2024-05-27","arxiv_id":"2405.16899","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-reinforcement-learning-5","title":"Provably Efficient Reinforcement Learning with Multinomial Logit Function Approximation","date":"2024-05-27","arxiv_id":"2405.17061","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-escape-route","title":"Reinforcement Learning Based Escape Route Generation in Low Visibility Environments","date":"2024-05-27","arxiv_id":"2406.07568","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cmdp-within-online-framework-for-meta-safe","title":"A CMDP-within-online framework for Meta-Safe Reinforcement Learning","date":"2024-05-26","arxiv_id":"2405.16601","repositories_listed":0,"syntology":null},{"url":null,"slug":"amortized-active-causal-induction-with-deep","title":"Amortized Active Causal Induction with Deep Reinforcement Learning","date":"2024-05-26","arxiv_id":"2405.16718","repositories_listed":0,"syntology":null},{"url":null,"slug":"make-safe-decisions-in-power-system-safe","title":"Make Safe Decisions in Power System: Safe Reinforcement Learning Based Pre-decision Making for Voltage Stability Emergency Control","date":"2024-05-26","arxiv_id":"2405.16485","repositories_listed":0,"syntology":null},{"url":null,"slug":"pick-up-the-pace-a-parameter-free-optimizer","title":"Fast TRAC: A Parameter-Free Optimizer for Lifelong Reinforcement Learning","date":"2024-05-26","arxiv_id":"2405.16642","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-jump-diffusions","title":"Reinforcement Learning for Jump-Diffusions, with Financial Applications","date":"2024-05-26","arxiv_id":"2405.16449","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlsf-reinforcement-learning-via-symbolic","title":"RLSF: Reinforcement Learning via Symbolic Feedback","date":"2024-05-26","arxiv_id":"2405.16661","repositories_listed":0,"syntology":null},{"url":"/paper/adaptive-q-network-on-the-fly-target","slug":"adaptive-q-network-on-the-fly-target","title":"Adaptive $Q$-Network: On-the-fly Target Selection for Deep Reinforcement Learning","date":"2024-05-25","arxiv_id":"2405.16195","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adaptive-q-network-on-the-fly-target#ran","syntology_url":"https://syntology.ai/paper/2405.16195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16195"}},"official":null}},{"url":null,"slug":"dynamic-inhomogeneous-quantum-resource","title":"Dynamic Inhomogeneous Quantum Resource Scheduling with Reinforcement Learning","date":"2024-05-25","arxiv_id":"2405.16380","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-analysis-for-conflict-avoidant","title":"Theoretical Study of Conflict-Avoidant Multi-Objective Reinforcement Learning","date":"2024-05-25","arxiv_id":"2405.16077","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-backdoor-attack-in-decentralized","title":"Cooperative Backdoor Attack in Decentralized Reinforcement Learning with Theoretical Guarantee","date":"2024-05-24","arxiv_id":"2405.15245","repositories_listed":0,"syntology":null},{"url":null,"slug":"counterexample-guided-repair-of-reinforcement","title":"Counterexample-Guided Repair of Reinforcement Learning Systems Using Safety Critics","date":"2024-05-24","arxiv_id":"2405.15430","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-via-large","title":"Extracting Heuristics from Large Language Models for Reward Shaping in Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15194","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-rlignment-inverse-reinforcement","title":"Inverse-RLignment: Large Language Model Alignment from Demonstrations through Inverse Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15624","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-generalizable-human-motion-generator","title":"Learning Generalizable Human Motion Generator with Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15541","repositories_listed":0,"syntology":null},{"url":null,"slug":"momentum-based-federated-reinforcement","title":"Momentum-Based Federated Reinforcement Learning with Interaction and Communication Efficiency","date":"2024-05-24","arxiv_id":"2405.17471","repositories_listed":0,"syntology":null},{"url":null,"slug":"trojanforge-adversarial-hardware-trojan","title":"TrojanForge: Generating Adversarial Hardware Trojan Examples Using Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15184","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-finite-time-analysis-of-distributed-q","title":"A finite time analysis of distributed Q-learning","date":"2024-05-23","arxiv_id":"2405.14078","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-5-5","title":"Deep Reinforcement Learning for 5*5 Multiplayer Go","date":"2024-05-23","arxiv_id":"2405.14265","repositories_listed":0,"syntology":null},{"url":null,"slug":"deterministic-policies-for-constrained","title":"Deterministic Policies for Constrained Reinforcement Learning in Polynomial Time","date":"2024-05-23","arxiv_id":"2405.14183","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-preference-optimization-with","title":"Direct Preference Optimization With Unobserved Preference Heterogeneity","date":"2024-05-23","arxiv_id":"2405.15065","repositories_listed":0,"syntology":null},{"url":null,"slug":"exclusively-penalized-q-learning-for-offline","title":"Exclusively Penalized Q-learning for Offline Reinforcement Learning","date":"2024-05-23","arxiv_id":"2405.14082","repositories_listed":0,"syntology":null},{"url":null,"slug":"privileged-sensing-scaffolds-reinforcement","title":"Privileged Sensing Scaffolds Reinforcement Learning","date":"2024-05-23","arxiv_id":"2405.14853","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-fine-tuning-text-1","title":"DLPO: Diffusion Model Loss-Guided Reinforcement Learning for Fine-Tuning Text-to-Speech Diffusion Models","date":"2024-05-23","arxiv_id":"2405.14632","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-constrained-offline-reinforcement","title":"State-Constrained Offline Reinforcement Learning","date":"2024-05-23","arxiv_id":"2405.14374","repositories_listed":0,"syntology":null},{"url":null,"slug":"almost-sure-convergence-rates-of-stochastic-1","title":"Almost sure convergence rates of stochastic gradient methods under gradient domination","date":"2024-05-22","arxiv_id":"2405.13592","repositories_listed":0,"syntology":null},{"url":null,"slug":"concertorl-an-innovative-time-interleaved","title":"ConcertoRL: An Innovative Time-Interleaved Reinforcement Learning Approach for Enhanced Control in Direct-Drive Tandem-Wing Vehicles","date":"2024-05-22","arxiv_id":"2405.13651","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-model-predictive-shielding-for","title":"Dynamic Model Predictive Shielding for Provably Safe Reinforcement Learning","date":"2024-05-22","arxiv_id":"2405.13863","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-search-advertising-strategies","title":"Optimizing Search Advertising Strategies: Integrating Reinforcement Learning with Generalized Second-Price Auctions for Enhanced Ad Ranking and Bidding","date":"2024-05-22","arxiv_id":"2405.13381","repositories_listed":0,"syntology":null},{"url":null,"slug":"traffic-control-using-intelligent-timing-of","title":"Traffic control using intelligent timing of traffic lights with reinforcement learning technique and real-time processing of surveillance camera images","date":"2024-05-22","arxiv_id":"2405.13256","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-learn-temporal-difference","title":"Transformers Learn Temporal Difference Methods for In-Context Reinforcement Learning","date":"2024-05-22","arxiv_id":"2405.13861","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-time-critical","title":"Deep Reinforcement Learning for Time-Critical Wilderness Search And Rescue Using Drones","date":"2024-05-21","arxiv_id":"2405.12800","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-linear-programming-framework-for","title":"A Unified Linear Programming Framework for Offline Reward Learning from Human Demonstrations and Feedback","date":"2024-05-20","arxiv_id":"2405.12421","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-deep-reinforcement-learning-for","title":"Continual Deep Reinforcement Learning for Decentralized Satellite Routing","date":"2024-05-20","arxiv_id":"2405.12308","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-punishment-reinforcement-learning-with","title":"Reward-Punishment Reinforcement Learning with Maximum Entropy","date":"2024-05-20","arxiv_id":"2405.11784","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparisons-are-all-you-need-for-optimizing","title":"Comparisons Are All You Need for Optimizing Smooth Functions","date":"2024-05-19","arxiv_id":"2405.11454","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-dive-into-model-free-reinforcement","title":"Deep Dive into Model-free Reinforcement Learning for Biological and Robotic Systems: Theory and Practice","date":"2024-05-19","arxiv_id":"2405.11457","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-no-harm-a-counterfactual-approach-to-safe","title":"Do No Harm: A Counterfactual Approach to Safe Reinforcement Learning","date":"2024-05-19","arxiv_id":"2405.11669","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-vehicle-aerodynamics-with-deep","title":"Enhancing Vehicle Aerodynamics with Deep Reinforcement Learning in Voxelised Models","date":"2024-05-19","arxiv_id":"2405.11492","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-distributional-value-functions-for","title":"Exploiting Distributional Value Functions for Financial Market Valuation, Enhanced Feature Creation and Improvement of Trading Algorithms","date":"2024-05-19","arxiv_id":"2405.11686","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-model-llm-for","title":"Large Language Model (LLM) for Telecommunications: A Comprehensive Survey on Principles, Key Techniques, and Opportunities","date":"2024-05-17","arxiv_id":"2405.10825","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-based-multi-agent-reinforcement-learning","title":"LLM-based Multi-Agent Reinforcement Learning: Current and Future Directions","date":"2024-05-17","arxiv_id":"2405.11106","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-constrained-reinforcement","title":"Sample-Efficient Constrained Reinforcement Learning with General Parameterization","date":"2024-05-17","arxiv_id":"2405.10624","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-varying-constraint-aware-reinforcement","title":"Time-Varying Constraint-Aware Reinforcement Learning for Energy Storage Control","date":"2024-05-17","arxiv_id":"2405.10536","repositories_listed":0,"syntology":null},{"url":null,"slug":"chaos-based-reinforcement-learning-with-td3","title":"Chaos-based reinforcement learning with TD3","date":"2024-05-15","arxiv_id":"2405.09086","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-continuous-integration-skip-a","title":"Detecting Continuous Integration Skip : A Reinforcement Learning-based Approach","date":"2024-05-15","arxiv_id":"2405.09657","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-distributed-fog-load-balancing-with","title":"Fully Distributed Fog Load Balancing with Multi-Agent Reinforcement Learning","date":"2024-05-15","arxiv_id":"2405.12236","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-resource-partitioning-on-modern","title":"Hierarchical Resource Partitioning on Modern GPUs: A Reinforcement Learning Approach","date":"2024-05-14","arxiv_id":"2405.08754","repositories_listed":0,"syntology":null},{"url":null,"slug":"i-ctrl-imitation-to-control-humanoid-robots","title":"I-CTRL: Imitation to Control Humanoid Robots Through Constrained Reinforcement Learning","date":"2024-05-14","arxiv_id":"2405.08726","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-deep-reinforcement-learning-for","title":"Optimizing Deep Reinforcement Learning for American Put Option Hedging","date":"2024-05-14","arxiv_id":"2405.08602","repositories_listed":0,"syntology":null},{"url":null,"slug":"python-based-reinforcement-learning-on","title":"Python-Based Reinforcement Learning on Simulink Models","date":"2024-05-14","arxiv_id":"2405.08567","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-constrained-multi-agent-reinforcement","title":"Safety Constrained Multi-Agent Reinforcement Learning for Active Voltage Control","date":"2024-05-14","arxiv_id":"2405.08443","repositories_listed":0,"syntology":null},{"url":null,"slug":"stable-inverse-reinforcement-learning","title":"Stable Inverse Reinforcement Learning: Policies from Control Lyapunov Landscapes","date":"2024-05-14","arxiv_id":"2405.08756","repositories_listed":0,"syntology":null},{"url":null,"slug":"hamiltonian-based-quantum-reinforcement","title":"Hamiltonian-based Quantum Reinforcement Learning for Neural Combinatorial Optimization","date":"2024-05-13","arxiv_id":"2405.07790","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-network-compression-for-reinforcement","title":"Neural Network Compression for Reinforcement Learning Tasks","date":"2024-05-13","arxiv_id":"2405.07748","repositories_listed":0,"syntology":null},{"url":null,"slug":"powqmix-weighted-value-factorization-with","title":"POWQMIX: Weighted Value Factorization with Potentially Optimal Joint Actions Recognition for Cooperative Multi-Agent Reinforcement Learning","date":"2024-05-13","arxiv_id":"2405.08036","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-risk-for-assistive-reinforcement","title":"Reducing Risk for Assistive Reinforcement Learning Policies with Diffusion Models","date":"2024-05-13","arxiv_id":"2405.07603","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-reinforcement-learning-for","title":"Structured Reinforcement Learning for Incentivized Stochastic Covert Optimization","date":"2024-05-13","arxiv_id":"2405.07415","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-demand-model-and-client-deployment-in","title":"On-Demand Model and Client Deployment in Federated Learning with Deep Reinforcement Learning","date":"2024-05-12","arxiv_id":"2405.07175","repositories_listed":0,"syntology":null},{"url":null,"slug":"auditing-an-automatic-grading-model-with-deep","title":"Auditing an Automatic Grading Model with deep Reinforcement Learning","date":"2024-05-11","arxiv_id":"2405.07087","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairness-in-reinforcement-learning-a-survey","title":"Fairness in Reinforcement Learning: A Survey","date":"2024-05-11","arxiv_id":"2405.06909","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-partial-survey-of-decentralized-cooperative","title":"An Initial Introduction to Cooperative Multi-Agent Reinforcement Learning","date":"2024-05-10","arxiv_id":"2405.06161","repositories_listed":0,"syntology":null},{"url":null,"slug":"hedging-american-put-options-with-deep","title":"Hedging American Put Options with Deep Reinforcement Learning","date":"2024-05-10","arxiv_id":"2405.06774","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-of-machine-learning-enabled-1","title":"An Overview of Machine Learning-Enabled Optimization for Reconfigurable Intelligent Surfaces-Aided 6G Networks: From Reinforcement Learning to Large Language Models","date":"2024-05-09","arxiv_id":"2405.17439","repositories_listed":0,"syntology":null},{"url":null,"slug":"conversational-topic-recommendation-in","title":"Conversational Topic Recommendation in Counseling and Psychotherapy with Decision Transformer and Large Language Models","date":"2024-05-08","arxiv_id":"2405.05060","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-stochastic-policy-gradient-negative","title":"Fast Stochastic Policy Gradient: Negative Momentum for Reinforcement Learning","date":"2024-05-08","arxiv_id":"2405.12228","repositories_listed":0,"syntology":null},{"url":null,"slug":"markowitz-meets-bellman-knowledge-distilled","title":"Markowitz Meets Bellman: Knowledge-distilled Reinforcement Learning for Portfolio Management","date":"2024-05-08","arxiv_id":"2405.05449","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-robust-ph-divergence-reinforcement","title":"Model-Free Robust $φ$-Divergence Reinforcement Learning Using Both Offline and Online Data","date":"2024-05-08","arxiv_id":"2405.05468","repositories_listed":0,"syntology":null}],"record_sha256":"8426d567ed31b94bb65977f674e7776e0adcc0fc9da61463c8444076d3a93089","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}