{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/55","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":55,"pages_in_order":132,"rows_per_page":100,"rows":[5401,5500],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/54","next":"/task/reinforcement-learning/papers/56","papers":[{"url":null,"slug":"state-constrained-offline-reinforcement","title":"State-Constrained Offline Reinforcement Learning","date":"2024-05-23","arxiv_id":"2405.14374","repositories_listed":0,"syntology":null},{"url":null,"slug":"almost-sure-convergence-rates-of-stochastic-1","title":"Almost sure convergence rates of stochastic gradient methods under gradient domination","date":"2024-05-22","arxiv_id":"2405.13592","repositories_listed":0,"syntology":null},{"url":null,"slug":"concertorl-an-innovative-time-interleaved","title":"ConcertoRL: An Innovative Time-Interleaved Reinforcement Learning Approach for Enhanced Control in Direct-Drive Tandem-Wing Vehicles","date":"2024-05-22","arxiv_id":"2405.13651","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-model-predictive-shielding-for","title":"Dynamic Model Predictive Shielding for Provably Safe Reinforcement Learning","date":"2024-05-22","arxiv_id":"2405.13863","repositories_listed":0,"syntology":null},{"url":null,"slug":"traffic-control-using-intelligent-timing-of","title":"Traffic control using intelligent timing of traffic lights with reinforcement learning technique and real-time processing of surveillance camera images","date":"2024-05-22","arxiv_id":"2405.13256","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-learn-temporal-difference","title":"Transformers Learn Temporal Difference Methods for In-Context Reinforcement Learning","date":"2024-05-22","arxiv_id":"2405.13861","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-time-critical","title":"Deep Reinforcement Learning for Time-Critical Wilderness Search And Rescue Using Drones","date":"2024-05-21","arxiv_id":"2405.12800","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-linear-programming-framework-for","title":"A Unified Linear Programming Framework for Offline Reward Learning from Human Demonstrations and Feedback","date":"2024-05-20","arxiv_id":"2405.12421","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-deep-reinforcement-learning-for","title":"Continual Deep Reinforcement Learning for Decentralized Satellite Routing","date":"2024-05-20","arxiv_id":"2405.12308","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-punishment-reinforcement-learning-with","title":"Reward-Punishment Reinforcement Learning with Maximum Entropy","date":"2024-05-20","arxiv_id":"2405.11784","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparisons-are-all-you-need-for-optimizing","title":"Comparisons Are All You Need for Optimizing Smooth Functions","date":"2024-05-19","arxiv_id":"2405.11454","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-dive-into-model-free-reinforcement","title":"Deep Dive into Model-free Reinforcement Learning for Biological and Robotic Systems: Theory and Practice","date":"2024-05-19","arxiv_id":"2405.11457","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-no-harm-a-counterfactual-approach-to-safe","title":"Do No Harm: A Counterfactual Approach to Safe Reinforcement Learning","date":"2024-05-19","arxiv_id":"2405.11669","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-vehicle-aerodynamics-with-deep","title":"Enhancing Vehicle Aerodynamics with Deep Reinforcement Learning in Voxelised Models","date":"2024-05-19","arxiv_id":"2405.11492","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-distributional-value-functions-for","title":"Exploiting Distributional Value Functions for Financial Market Valuation, Enhanced Feature Creation and Improvement of Trading Algorithms","date":"2024-05-19","arxiv_id":"2405.11686","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-model-llm-for","title":"Large Language Model (LLM) for Telecommunications: A Comprehensive Survey on Principles, Key Techniques, and Opportunities","date":"2024-05-17","arxiv_id":"2405.10825","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-based-multi-agent-reinforcement-learning","title":"LLM-based Multi-Agent Reinforcement Learning: Current and Future Directions","date":"2024-05-17","arxiv_id":"2405.11106","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-constrained-reinforcement","title":"Sample-Efficient Constrained Reinforcement Learning with General Parameterization","date":"2024-05-17","arxiv_id":"2405.10624","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-varying-constraint-aware-reinforcement","title":"Time-Varying Constraint-Aware Reinforcement Learning for Energy Storage Control","date":"2024-05-17","arxiv_id":"2405.10536","repositories_listed":0,"syntology":null},{"url":null,"slug":"chaos-based-reinforcement-learning-with-td3","title":"Chaos-based reinforcement learning with TD3","date":"2024-05-15","arxiv_id":"2405.09086","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-continuous-integration-skip-a","title":"Detecting Continuous Integration Skip : A Reinforcement Learning-based Approach","date":"2024-05-15","arxiv_id":"2405.09657","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-distributed-fog-load-balancing-with","title":"Fully Distributed Fog Load Balancing with Multi-Agent Reinforcement Learning","date":"2024-05-15","arxiv_id":"2405.12236","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-resource-partitioning-on-modern","title":"Hierarchical Resource Partitioning on Modern GPUs: A Reinforcement Learning Approach","date":"2024-05-14","arxiv_id":"2405.08754","repositories_listed":0,"syntology":null},{"url":null,"slug":"i-ctrl-imitation-to-control-humanoid-robots","title":"I-CTRL: Imitation to Control Humanoid Robots Through Constrained Reinforcement Learning","date":"2024-05-14","arxiv_id":"2405.08726","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-deep-reinforcement-learning-for","title":"Optimizing Deep Reinforcement Learning for American Put Option Hedging","date":"2024-05-14","arxiv_id":"2405.08602","repositories_listed":0,"syntology":null},{"url":null,"slug":"python-based-reinforcement-learning-on","title":"Python-Based Reinforcement Learning on Simulink Models","date":"2024-05-14","arxiv_id":"2405.08567","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-constrained-multi-agent-reinforcement","title":"Safety Constrained Multi-Agent Reinforcement Learning for Active Voltage Control","date":"2024-05-14","arxiv_id":"2405.08443","repositories_listed":0,"syntology":null},{"url":null,"slug":"stable-inverse-reinforcement-learning","title":"Stable Inverse Reinforcement Learning: Policies from Control Lyapunov Landscapes","date":"2024-05-14","arxiv_id":"2405.08756","repositories_listed":0,"syntology":null},{"url":null,"slug":"hamiltonian-based-quantum-reinforcement","title":"Hamiltonian-based Quantum Reinforcement Learning for Neural Combinatorial Optimization","date":"2024-05-13","arxiv_id":"2405.07790","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-network-compression-for-reinforcement","title":"Neural Network Compression for Reinforcement Learning Tasks","date":"2024-05-13","arxiv_id":"2405.07748","repositories_listed":0,"syntology":null},{"url":null,"slug":"powqmix-weighted-value-factorization-with","title":"POWQMIX: Weighted Value Factorization with Potentially Optimal Joint Actions Recognition for Cooperative Multi-Agent Reinforcement Learning","date":"2024-05-13","arxiv_id":"2405.08036","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-risk-for-assistive-reinforcement","title":"Reducing Risk for Assistive Reinforcement Learning Policies with Diffusion Models","date":"2024-05-13","arxiv_id":"2405.07603","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-reinforcement-learning-for","title":"Structured Reinforcement Learning for Incentivized Stochastic Covert Optimization","date":"2024-05-13","arxiv_id":"2405.07415","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-demand-model-and-client-deployment-in","title":"On-Demand Model and Client Deployment in Federated Learning with Deep Reinforcement Learning","date":"2024-05-12","arxiv_id":"2405.07175","repositories_listed":0,"syntology":null},{"url":null,"slug":"auditing-an-automatic-grading-model-with-deep","title":"Auditing an Automatic Grading Model with deep Reinforcement Learning","date":"2024-05-11","arxiv_id":"2405.07087","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairness-in-reinforcement-learning-a-survey","title":"Fairness in Reinforcement Learning: A Survey","date":"2024-05-11","arxiv_id":"2405.06909","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-partial-survey-of-decentralized-cooperative","title":"An Initial Introduction to Cooperative Multi-Agent Reinforcement Learning","date":"2024-05-10","arxiv_id":"2405.06161","repositories_listed":0,"syntology":null},{"url":null,"slug":"hedging-american-put-options-with-deep","title":"Hedging American Put Options with Deep Reinforcement Learning","date":"2024-05-10","arxiv_id":"2405.06774","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-of-machine-learning-enabled-1","title":"An Overview of Machine Learning-Enabled Optimization for Reconfigurable Intelligent Surfaces-Aided 6G Networks: From Reinforcement Learning to Large Language Models","date":"2024-05-09","arxiv_id":"2405.17439","repositories_listed":0,"syntology":null},{"url":null,"slug":"conversational-topic-recommendation-in","title":"Conversational Topic Recommendation in Counseling and Psychotherapy with Decision Transformer and Large Language Models","date":"2024-05-08","arxiv_id":"2405.05060","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-stochastic-policy-gradient-negative","title":"Fast Stochastic Policy Gradient: Negative Momentum for Reinforcement Learning","date":"2024-05-08","arxiv_id":"2405.12228","repositories_listed":0,"syntology":null},{"url":null,"slug":"markowitz-meets-bellman-knowledge-distilled","title":"Markowitz Meets Bellman: Knowledge-distilled Reinforcement Learning for Portfolio Management","date":"2024-05-08","arxiv_id":"2405.05449","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-robust-ph-divergence-reinforcement","title":"Model-Free Robust $φ$-Divergence Reinforcement Learning Using Both Offline and Online Data","date":"2024-05-08","arxiv_id":"2405.05468","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-offline-reinforcement-learning-with","title":"Improving Offline Reinforcement Learning with Inaccurate Simulators","date":"2024-05-07","arxiv_id":"2405.04307","repositories_listed":0,"syntology":null},{"url":null,"slug":"racer-epistemic-risk-sensitive-rl-enables","title":"RACER: Epistemic Risk-Sensitive RL Enables Fast Driving with Fewer Crashes","date":"2024-05-07","arxiv_id":"2405.04714","repositories_listed":0,"syntology":null},{"url":null,"slug":"torchdriveenv-a-reinforcement-learning","title":"TorchDriveEnv: A Reinforcement Learning Benchmark for Autonomous Driving with Reactive, Realistic, and Diverse Non-Playable Characters","date":"2024-05-07","arxiv_id":"2405.04491","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-reinforcement-learning-with-1","title":"Federated Reinforcement Learning with Constraint Heterogeneity","date":"2024-05-06","arxiv_id":"2405.03236","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-convergence-and-sample-complexity-1","title":"Finite-Time Convergence and Sample Complexity of Actor-Critic Multi-Objective Reinforcement Learning","date":"2024-05-05","arxiv_id":"2405.03082","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-learned-non","title":"Safe Reinforcement Learning with Learned Non-Markovian Safety Constraints","date":"2024-05-05","arxiv_id":"2405.03005","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-safe-set-algorithm-for-provably-safe","title":"Implicit Safe Set Algorithm for Provably Safe Reinforcement Learning","date":"2024-05-04","arxiv_id":"2405.02754","repositories_listed":0,"syntology":null},{"url":null,"slug":"uduc-an-uncertainty-driven-approach-for","title":"UDUC: An Uncertainty-driven Approach for Learning-based Robust Control","date":"2024-05-04","arxiv_id":"2405.02598","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-robot-soccer-from-egocentric-vision","title":"Learning Robot Soccer from Egocentric Vision with Deep Reinforcement Learning","date":"2024-05-03","arxiv_id":"2405.02425","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-8","title":"Model-based reinforcement learning for protein backbone design","date":"2024-05-03","arxiv_id":"2405.01983","repositories_listed":0,"syntology":null},{"url":null,"slug":"socialgfs-learning-social-gradient-fields-for","title":"SocialGFs: Learning Social Gradient Fields for Multi-Agent Reinforcement Learning","date":"2024-05-03","arxiv_id":"2405.01839","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-sum-positional-differential-games-as-a","title":"Zero-Sum Positional Differential Games as a Framework for Robust Reinforcement Learning: Deep Q-Learning Approach","date":"2024-05-03","arxiv_id":"2405.02044","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-imitation-for-manipulator-control","title":"Behavior Imitation for Manipulator Control and Grasping with Deep Reinforcement Learning","date":"2024-05-02","arxiv_id":"2405.01284","repositories_listed":0,"syntology":null},{"url":null,"slug":"citylearn-v2-energy-flexible-resilient","title":"CityLearn v2: Energy-flexible, resilient, occupant-centric, and carbon-aware management of grid-interactive communities","date":"2024-05-02","arxiv_id":"2405.03848","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-under","title":"Constrained Reinforcement Learning Under Model Mismatch","date":"2024-05-02","arxiv_id":"2405.01327","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-conditioned-reinforcement-learning-for","title":"Goal-conditioned reinforcement learning for ultrasound navigation guidance","date":"2024-05-02","arxiv_id":"2405.01409","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-hybrid-resource-allocation-in-mec","title":"Intelligent Hybrid Resource Allocation in MEC-assisted RAN Slicing Network","date":"2024-05-02","arxiv_id":"2405.17436","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-guided-semi-supervised","title":"Reinforcement Learning-Guided Semi-Supervised Learning","date":"2024-05-02","arxiv_id":"2405.01760","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-risk-sensitive-reinforcement-learning-1","title":"Robust Risk-Sensitive Reinforcement Learning with Conditional Value-at-Risk","date":"2024-05-02","arxiv_id":"2405.01718","repositories_listed":0,"syntology":null},{"url":null,"slug":"tabular-and-deep-reinforcement-learning-for","title":"Tabular and Deep Reinforcement Learning for Gittins Index","date":"2024-05-02","arxiv_id":"2405.01157","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-interpretable-reinforcement-learning-1","title":"Towards Interpretable Reinforcement Learning with Constrained Normalizing Flow Policies","date":"2024-05-02","arxiv_id":"2405.01198","repositories_listed":0,"syntology":null},{"url":null,"slug":"employing-federated-learning-for-training","title":"Employing Federated Learning for Training Autonomous HVAC Systems","date":"2024-05-01","arxiv_id":"2405.00389","repositories_listed":0,"syntology":null},{"url":null,"slug":"mf-oml-online-mean-field-reinforcement","title":"MF-OML: Online Mean-Field Reinforcement Learning with Occupation Measures for Large Population Games","date":"2024-05-01","arxiv_id":"2405.00282","repositories_listed":0,"syntology":null},{"url":null,"slug":"portfolio-management-using-deep-reinforcement","title":"Portfolio Management using Deep Reinforcement Learning","date":"2024-05-01","arxiv_id":"2405.01604","repositories_listed":0,"syntology":null},{"url":null,"slug":"queue-based-eco-driving-at-roundabouts-with","title":"Queue-based Eco-Driving at Roundabouts with Reinforcement Learning","date":"2024-05-01","arxiv_id":"2405.00625","repositories_listed":0,"syntology":null},{"url":null,"slug":"bias-mitigation-via-compensation-a","title":"AI, Pluralism, and (Social) Compensation","date":"2024-04-30","arxiv_id":"2404.19256","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-problem-solving-with","title":"Reinforcement Learning Problem Solving with Large Language Models","date":"2024-04-29","arxiv_id":"2404.18638","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-training-superconducting-neuromorphic","title":"Self-training superconducting neuromorphic circuits using reinforcement learning rules","date":"2024-04-29","arxiv_id":"2404.18774","repositories_listed":0,"syntology":null},{"url":null,"slug":"verco-learning-coordinated-verbal","title":"Verco: Learning Coordinated Verbal Communication for Multi-agent Reinforcement Learning","date":"2024-04-27","arxiv_id":"2404.17780","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-transfer-for-cross-domain","title":"Knowledge Transfer for Cross-Domain Reinforcement Learning: A Systematic Review","date":"2024-04-26","arxiv_id":"2404.17687","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-8","title":"Offline Reinforcement Learning with Behavioral Supervisor Tuning","date":"2024-04-25","arxiv_id":"2404.16399","repositories_listed":0,"syntology":null},{"url":"/paper/dpo-differential-reinforcement-learning-with","slug":"dpo-differential-reinforcement-learning-with","title":"DPO: A Differential and Pointwise Control Approach to Reinforcement Learning","date":"2024-04-24","arxiv_id":"2404.15617","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/dpo-differential-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2404.15617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15617"}},"official":null}},{"url":null,"slug":"grsn-gated-recurrent-spiking-neurons-for","title":"GRSN: Gated Recurrent Spiking Neurons for POMDPs and MARL","date":"2024-04-24","arxiv_id":"2404.15597","repositories_listed":0,"syntology":null},{"url":null,"slug":"cache-aware-reinforcement-learning-in-large","title":"Cache-Aware Reinforcement Learning in Large-Scale Recommender Systems","date":"2024-04-23","arxiv_id":"2404.14961","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-high-speed-cruising-performance-of","title":"Enhancing High-Speed Cruising Performance of Autonomous Vehicles through Integrated Deep Reinforcement Learning Framework","date":"2024-04-23","arxiv_id":"2404.14713","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-reinforcement-learning-via-1","title":"Evolutionary Reinforcement Learning via Cooperative Coevolution","date":"2024-04-23","arxiv_id":"2404.14763","repositories_listed":0,"syntology":null},{"url":null,"slug":"multistop-solving-functional-equations-with","title":"MultiSTOP: Solving Functional Equations with Reinforcement Learning","date":"2024-04-23","arxiv_id":"2404.14909","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-power-of-resets-in-online-reinforcement","title":"The Power of Resets in Online Reinforcement Learning","date":"2024-04-23","arxiv_id":"2404.15417","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-deep-reinforcement-learning-to-promote","title":"Using deep reinforcement learning to promote sustainable human behaviour on a common pool resource problem","date":"2024-04-23","arxiv_id":"2404.15059","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-control-barrier-functions-and-their","title":"Learning Control Barrier Functions and their application in Reinforcement Learning: A Survey","date":"2024-04-22","arxiv_id":"2404.16879","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-stability-of-lipschitz-continuous","title":"On the stability of Lipschitz continuous control problems and its application to reinforcement learning","date":"2024-04-20","arxiv_id":"2404.13316","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-time-risk-sensitive-reinforcement","title":"Continuous-time Risk-sensitive Reinforcement Learning via Quadratic Variation Penalty","date":"2024-04-19","arxiv_id":"2404.12598","repositories_listed":0,"syntology":null},{"url":null,"slug":"mm-phyrlhf-reinforcement-learning-framework","title":"MM-PhyRLHF: Reinforcement Learning Framework for Multimodal Physics Question-Answering","date":"2024-04-19","arxiv_id":"2404.12926","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approach-for-1","title":"Reinforcement Learning Approach for Integrating Compressed Contexts into Knowledge Graphs","date":"2024-04-19","arxiv_id":"2404.12587","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-task-continual-offline-reinforcement","title":"Data-Incremental Continual Offline Reinforcement Learning","date":"2024-04-19","arxiv_id":"2404.12639","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-stitching-in-reinforcement-learning","title":"Zero-Shot Stitching in Reinforcement Learning using Relative Representations","date":"2024-04-19","arxiv_id":"2404.12917","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-reinforcement-learning-with-1","title":"Actor-Critic Reinforcement Learning with Phased Actor","date":"2024-04-18","arxiv_id":"2404.11834","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-policy-optimization-with-temporal-logic","title":"LTL-Constrained Policy Optimization with Cycle Experience Replay","date":"2024-04-17","arxiv_id":"2404.11578","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-re-calibration-of-quantum-devices","title":"Automatic re-calibration of quantum devices by reinforcement learning","date":"2024-04-16","arxiv_id":"2404.10726","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-control-reinforcement-learning","title":"Continuous Control Reinforcement Learning: Distributed Distributional DrQ Algorithms","date":"2024-04-16","arxiv_id":"2404.10645","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-trajectory-generalization-for-offline","title":"Offline Trajectory Generalization for Offline Reinforcement Learning","date":"2024-04-16","arxiv_id":"2404.10393","repositories_listed":0,"syntology":null},{"url":"/paper/randomized-exploration-in-cooperative-multi","slug":"randomized-exploration-in-cooperative-multi","title":"Randomized Exploration in Cooperative Multi-Agent Reinforcement Learning","date":"2024-04-16","arxiv_id":"2404.10728","repositories_listed":0,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/randomized-exploration-in-cooperative-multi#ran","syntology_url":"https://syntology.ai/paper/2404.10728","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10728"}},"official":null}},{"url":null,"slug":"simplex-decomposition-for-portfolio","title":"Simplex Decomposition for Portfolio Allocation Constraints in Reinforcement Learning","date":"2024-04-16","arxiv_id":"2404.10683","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-research-community-in-interpretable","title":"Towards a Research Community in Interpretable Reinforcement Learning: the InterpPol Workshop","date":"2024-04-16","arxiv_id":"2404.10906","repositories_listed":0,"syntology":null},{"url":null,"slug":"deceiving-to-enlighten-coaxing-llms-to-self","title":"Reinforcement Learning from Multi-role Debates as Feedback for Bias Mitigation in LLMs","date":"2024-04-15","arxiv_id":"2404.10160","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-reinforcement-learning-based-on","title":"Effective Reinforcement Learning Based on Structural Information Principles","date":"2024-04-15","arxiv_id":"2404.09760","repositories_listed":0,"syntology":null},{"url":null,"slug":"eyeformer-predicting-personalized-scanpaths","title":"EyeFormer: Predicting Personalized Scanpaths with Transformer-Guided Reinforcement Learning","date":"2024-04-15","arxiv_id":"2404.10163","repositories_listed":0,"syntology":null}],"record_sha256":"4d65cd48a7fe17363447e135cf7910c6cc1077b36b880d3d1fd69defbf0cb84f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}