{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/63","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":63,"pages_in_order":135,"rows_per_page":100,"rows":[6201,6300],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/62","next":"/task/reinforcement-learning-2/papers/64","papers":[{"url":null,"slug":"graph-enabled-reinforcement-learning-for-time","title":"Graph-enabled Reinforcement Learning for Time Series Forecasting with Adaptive Intelligence","date":"2023-09-18","arxiv_id":"2309.10186","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-online-distillation-promoting-safe","title":"Guided Online Distillation: Promoting Safe Reinforcement Learning by Offline Demonstration","date":"2023-09-18","arxiv_id":"2309.09408","repositories_listed":0,"syntology":null},{"url":null,"slug":"mechanic-maker-2-0-reinforcement-learning-for","title":"Mechanic Maker 2.0: Reinforcement Learning for Evaluating Generated Rules","date":"2023-09-18","arxiv_id":"2309.09476","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-14","title":"Multi-Agent Deep Reinforcement Learning for Cooperative and Competitive Autonomous Vehicles using AutoDRIVE Ecosystem","date":"2023-09-18","arxiv_id":"2309.10007","repositories_listed":0,"syntology":null},{"url":null,"slug":"privileged-to-predicted-towards-sensorimotor","title":"Privileged to Predicted: Towards Sensorimotor Reinforcement Learning for Urban Driving","date":"2023-09-18","arxiv_id":"2309.09756","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-transformer-scalable-offline-reinforcement","title":"Q-Transformer: Scalable Offline Reinforcement Learning via Autoregressive Q-Functions","date":"2023-09-18","arxiv_id":"2309.10150","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-deep-reinforcement-learning-with","title":"Sim-to-Real Deep Reinforcement Learning with Manipulators for Pick-and-place","date":"2023-09-17","arxiv_id":"2309.09247","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-mildly-conservative-model-based","title":"DOMAIN: MilDly COnservative Model-BAsed OfflINe Reinforcement Learning","date":"2023-09-16","arxiv_id":"2309.08925","repositories_listed":0,"syntology":null},{"url":null,"slug":"gym-saturation-gymnasium-environments-for","title":"gym-saturation: Gymnasium environments for saturation provers (System description)","date":"2023-09-16","arxiv_id":"2309.09022","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-efficient-and","title":"Deep Reinforcement Learning for Efficient and Fair Allocation of Health Care Resources","date":"2023-09-15","arxiv_id":"2309.08560","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantitative-and-qualitative-evaluation-of-1","title":"Autonomous and Human-Driven Vehicles Interacting in a Roundabout: A Quantitative and Qualitative Evaluation","date":"2023-09-15","arxiv_id":"2309.08254","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-multi-agent-reinforcement-learning-for-3","title":"Deep Multi-Agent Reinforcement Learning for Decentralized Active Hypothesis Testing","date":"2023-09-14","arxiv_id":"2309.08477","repositories_listed":0,"syntology":null},{"url":null,"slug":"equivariant-data-augmentation-for","title":"Equivariant Data Augmentation for Generalization in Offline Reinforcement Learning","date":"2023-09-14","arxiv_id":"2309.07578","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-space-abstraction-in-hierarchical-1","title":"Goal Space Abstraction in Hierarchical Reinforcement Learning via Set-Based Reachability Analysis","date":"2023-09-14","arxiv_id":"2309.07675","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-real-world-quadrupedal-locomotion-benchmark","title":"A Real-World Quadrupedal Locomotion Benchmark for Offline Reinforcement Learning","date":"2023-09-13","arxiv_id":"2309.16718","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-loss-adjusted-prioritized","title":"Attention Loss Adjusted Prioritized Experience Replay","date":"2023-09-13","arxiv_id":"2309.06684","repositories_listed":0,"syntology":null},{"url":null,"slug":"characterizing-speed-performance-of-multi","title":"Characterizing Speed Performance of Multi-Agent Reinforcement Learning","date":"2023-09-13","arxiv_id":"2309.07108","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-quantum-recurrent-reinforcement","title":"Efficient quantum recurrent reinforcement learning via quantum reservoir computing","date":"2023-09-13","arxiv_id":"2309.07339","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-impact-of-action","title":"Investigating the Impact of Action Representations in Policy Gradient Algorithms","date":"2023-09-13","arxiv_id":"2309.06921","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-engineered-value-decomposition","title":"Privacy-Engineered Value Decomposition Networks for Cooperative Multi-Agent Reinforcement Learning","date":"2023-09-13","arxiv_id":"2311.06255","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-dual","title":"Safe Reinforcement Learning with Dual Robustness","date":"2023-09-13","arxiv_id":"2309.06835","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-communication-in-multi-agent","title":"Emergent Communication in Multi-Agent Reinforcement Learning for Future Wireless Networks","date":"2023-09-12","arxiv_id":"2309.06021","repositories_listed":0,"syntology":null},{"url":null,"slug":"fidelity-induced-interpretable-policy","title":"Fidelity-Induced Interpretable Policy Extraction for Reinforcement Learning","date":"2023-09-12","arxiv_id":"2309.06097","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-space-abstraction-in-hierarchical","title":"Goal Space Abstraction in Hierarchical Reinforcement Learning via Reachability Analysis","date":"2023-09-12","arxiv_id":"2309.07168","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-topological-operations-on-meshes","title":"Learning topological operations on meshes with application to block decomposition of polygons","date":"2023-09-12","arxiv_id":"2309.06484","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-aware-reinforcement-learning-through","title":"Risk-Aware Reinforcement Learning through Optimal Transport Theory","date":"2023-09-12","arxiv_id":"2309.06239","repositories_listed":0,"syntology":null},{"url":null,"slug":"career-path-recommendations-for-long-term","title":"Career Path Recommendations for Long-term Income Maximization: A Reinforcement Learning Approach","date":"2023-09-11","arxiv_id":"2309.05391","repositories_listed":0,"syntology":null},{"url":null,"slug":"mind-the-uncertainty-risk-aware-and-actively","title":"Mind the Uncertainty: Risk-Aware and Actively Exploring Model-Based Reinforcement Learning","date":"2023-09-11","arxiv_id":"2309.05582","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-informed-reinforcement-learning-via","title":"Physics-informed reinforcement learning via probabilistic co-adjustment functions","date":"2023-09-11","arxiv_id":"2309.05404","repositories_listed":0,"syntology":null},{"url":null,"slug":"chasing-the-intruder-a-reinforcement-learning","title":"Chasing the Intruder: A Reinforcement Learning Approach for Tracking Intruder Drones","date":"2023-09-10","arxiv_id":"2309.05070","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-learning-in-low-rank-slate","title":"Representation Learning in Low-rank Slate-based Recommender Systems","date":"2023-09-10","arxiv_id":"2309.08622","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-influencers-in-complex-networks-an","title":"Finding Influencers in Complex Networks: An Effective Deep Reinforcement Learning Approach","date":"2023-09-09","arxiv_id":"2309.07153","repositories_listed":0,"syntology":null},{"url":null,"slug":"verifiable-reinforcement-learning-systems-via","title":"Verifiable Reinforcement Learning Systems via Compositionality","date":"2023-09-09","arxiv_id":"2309.06420","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-world-model-disentanglement-in","title":"Leveraging World Model Disentanglement in Value-Based Multi-Agent Reinforcement Learning","date":"2023-09-08","arxiv_id":"2309.04615","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapping-adaptive-human-machine","title":"Bootstrapping Adaptive Human-Machine Interfaces with Offline Reinforcement Learning","date":"2023-09-07","arxiv_id":"2309.03839","repositories_listed":0,"syntology":null},{"url":null,"slug":"chat-failures-and-troubles-reasons-and","title":"Chat Failures and Troubles: Reasons and Solutions","date":"2023-09-07","arxiv_id":"2309.03708","repositories_listed":0,"syntology":null},{"url":null,"slug":"marketing-budget-allocation-with-offline","title":"Marketing Budget Allocation with Offline Constrained Deep Reinforcement Learning","date":"2023-09-06","arxiv_id":"2309.02669","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-continuous-time-reinforcement-learning","title":"Near-continuous time Reinforcement Learning for continuous state-action spaces","date":"2023-09-06","arxiv_id":"2309.02815","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-reducing-undesirable-behavior-in-deep","title":"On Reducing Undesirable Behavior in Deep Reinforcement Learning Models","date":"2023-09-06","arxiv_id":"2309.02869","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-physics-informed-reinforcement","title":"A Survey on Physics Informed Reinforcement Learning: Review and Open Problems","date":"2023-09-05","arxiv_id":"2309.01909","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributionally-robust-model-based","title":"Distributionally Robust Model-based Reinforcement Learning with Large State Spaces","date":"2023-09-05","arxiv_id":"2309.02236","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-federated-deep-reinforcement","title":"Personalized Federated Deep Reinforcement Learning-based Trajectory Optimization for Multi-UAV Assisted Edge Computing","date":"2023-09-05","arxiv_id":"2309.02193","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-flow-control-for-three-dimensional","title":"Active flow control for three-dimensional cylinders through deep reinforcement learning","date":"2023-09-04","arxiv_id":"2309.02462","repositories_listed":0,"syntology":null},{"url":null,"slug":"hawkeye-change-targeted-testing-for-android","title":"Hawkeye: Change-targeted Testing for Android Apps based on Deep Reinforcement Learning","date":"2023-09-04","arxiv_id":"2309.01519","repositories_listed":0,"syntology":null},{"url":null,"slug":"hundreds-guide-millions-adaptive-offline","title":"Hundreds Guide Millions: Adaptive Offline Reinforcement Learning with Expert Guidance","date":"2023-09-04","arxiv_id":"2309.01448","repositories_listed":0,"syntology":null},{"url":null,"slug":"looptune-optimizing-tensor-computations-with","title":"LoopTune: Optimizing Tensor Computations with Reinforcement Learning","date":"2023-09-04","arxiv_id":"2309.01825","repositories_listed":0,"syntology":null},{"url":null,"slug":"neurosymbolic-reinforcement-learning-and","title":"Neurosymbolic Reinforcement Learning and Planning: A Survey","date":"2023-09-02","arxiv_id":"2309.01038","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-of-deep-learning-methods-in","title":"Application of Deep Learning Methods in Monitoring and Optimization of Electric Power Systems","date":"2023-09-01","arxiv_id":"2309.00498","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-lidar-driven-reinforcement","title":"End-to-end Lidar-Driven Reinforcement Learning for Autonomous Racing","date":"2023-09-01","arxiv_id":"2309.00296","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-human-feedback-1","title":"Reinforcement Learning with Human Feedback for Realistic Traffic Simulation","date":"2023-09-01","arxiv_id":"2309.00709","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlaif-scaling-reinforcement-learning-from","title":"RLAIF vs. RLHF: Scaling Reinforcement Learning from Human Feedback with AI Feedback","date":"2023-09-01","arxiv_id":"2309.00267","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-policy-adaptation-method-for-implicit","title":"Foundational Policy Acquisition via Multitask Learning for Motor Skill Generation","date":"2023-08-31","arxiv_id":"2308.16471","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-decision-transformers-for","title":"Multi-Objective Decision Transformers for Offline Reinforcement Learning","date":"2023-08-31","arxiv_id":"2308.16379","repositories_listed":0,"syntology":null},{"url":null,"slug":"cyclophobic-reinforcement-learning","title":"Cyclophobic Reinforcement Learning","date":"2023-08-30","arxiv_id":"2308.15911","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-inductive-logic-programming-meets","title":"Deep Inductive Logic Programming meets Reinforcement Learning","date":"2023-08-30","arxiv_id":"2308.16210","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-style-transfer-for-robust-policy-1","title":"Adversarial Style Transfer for Robust Policy Optimization in Deep Reinforcement Learning","date":"2023-08-29","arxiv_id":"2308.15550","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-framework","title":"Deep Reinforcement Learning Based Framework for Mobile Energy Disseminator Dispatching to Charge On-the-Road Electric Vehicles","date":"2023-08-29","arxiv_id":"2308.15656","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-multi-agent-target-search-and","title":"Distributed multi-agent target search and tracking with Gaussian process and reinforcement learning","date":"2023-08-29","arxiv_id":"2308.14971","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-composition-in-reinforcement-learning","title":"Policy composition in reinforcement learning via multi-objective policy optimization","date":"2023-08-29","arxiv_id":"2308.15470","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-uplink","title":"Deep Reinforcement Learning for Uplink Scheduling in NOMA-URLLC Networks","date":"2023-08-28","arxiv_id":"2308.14523","repositories_listed":0,"syntology":null},{"url":null,"slug":"maneuver-decision-making-through-proximal","title":"Maneuver Decision-Making Through Proximal Policy Optimization And Monte Carlo Tree Search","date":"2023-08-28","arxiv_id":"2309.08611","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-reward-structures-of-markov-decision","title":"On Reward Structures of Markov Decision Processes","date":"2023-08-28","arxiv_id":"2308.14919","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-economic-gas-turbine-dispatch-with","title":"Optimal Economic Gas Turbine Dispatch with Deep Reinforcement Learning","date":"2023-08-28","arxiv_id":"2308.14924","repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-progress-in-energy-management-of","title":"Recent Progress in Energy Management of Connected Hybrid Electric Vehicles Using Reinforcement Learning","date":"2023-08-28","arxiv_id":"2308.14602","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-generative-ai-a","title":"Reinforcement Learning for Generative AI: A Survey","date":"2023-08-28","arxiv_id":"2308.14328","repositories_listed":0,"syntology":null},{"url":null,"slug":"spread-control-method-on-unknown-networks","title":"Spread Control Method on Unknown Networks Based on Hierarchical Reinforcement Learning","date":"2023-08-28","arxiv_id":"2308.14311","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistically-efficient-variance-reduction","title":"Statistically Efficient Variance Reduction with Double Policy Estimation for Off-Policy Evaluation in Sequence-Modeled Reinforcement Learning","date":"2023-08-28","arxiv_id":"2308.14897","repositories_listed":0,"syntology":null},{"url":null,"slug":"target-independent-xla-optimization-using","title":"Target-independent XLA optimization using Reinforcement Learning","date":"2023-08-28","arxiv_id":"2308.14364","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-optimal-control-2","title":"Reinforcement Learning-based Optimal Control and Software Rejuvenation for Safe and Efficient UAV Navigation","date":"2023-08-27","arxiv_id":"2308.14139","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-usage-of-qubo-based","title":"A Graph Neural Network-Based QUBO-Formulated Hamiltonian-Inspired Loss Function for Combinatorial Optimization using Reinforcement Learning","date":"2023-08-27","arxiv_id":"2308.13978","repositories_listed":0,"syntology":null},{"url":null,"slug":"jax-lob-a-gpu-accelerated-limit-order-book","title":"JAX-LOB: A GPU-Accelerated limit order book simulator to unlock large scale reinforcement learning for trading","date":"2023-08-25","arxiv_id":"2308.13289","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-cyber-defence-tactics-from-scratch","title":"Learning Cyber Defence Tactics from Scratch with Multi-Agent Reinforcement Learning","date":"2023-08-25","arxiv_id":"2310.05939","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-knowledge-and-reinforcement","title":"Leveraging Knowledge and Reinforcement Learning for Enhanced Reliability of Language Models","date":"2023-08-25","arxiv_id":"2308.13467","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-reinforcement-learning-with","title":"Model-free Reinforcement Learning with Stochastic Reward Stabilization for Recommender Systems","date":"2023-08-25","arxiv_id":"2308.13246","repositories_listed":0,"syntology":null},{"url":null,"slug":"nonparametric-additive-value-functions","title":"Nonparametric Additive Value Functions: Interpretable Reinforcement Learning with an Application to Surgical Recovery","date":"2023-08-25","arxiv_id":"2308.13135","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-assisted-evolutionary","title":"Reinforcement Learning-assisted Evolutionary Algorithm: A Survey and Research Opportunities","date":"2023-08-25","arxiv_id":"2308.13420","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-optimal-head-to-head-autonomous","title":"Towards Optimal Head-to-head Autonomous Racing with Curriculum Reinforcement Learning","date":"2023-08-25","arxiv_id":"2308.13491","repositories_listed":0,"syntology":null},{"url":null,"slug":"actuator-trajectory-planning-for-uavs-with","title":"Actuator Trajectory Planning for UAVs with Overhead Manipulator using Reinforcement Learning","date":"2023-08-24","arxiv_id":"2308.12843","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-efficient-distributed-multi-agent","title":"An Efficient Distributed Multi-Agent Reinforcement Learning for EV Charging Network Control","date":"2023-08-24","arxiv_id":"2308.12921","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditional-kernel-imitation-learning-for","title":"Conditional Kernel Imitation Learning for Continuous State Environments","date":"2023-08-24","arxiv_id":"2308.12573","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-driven-cross","title":"Deep Reinforcement Learning-driven Cross-Community Energy Interaction Optimal Scheduling","date":"2023-08-24","arxiv_id":"2308.12554","repositories_listed":0,"syntology":null},{"url":null,"slug":"extreme-risk-mitigation-in-reinforcement","title":"Extreme Risk Mitigation in Reinforcement Learning using Extreme Value Theory","date":"2023-08-24","arxiv_id":"2308.13011","repositories_listed":0,"syntology":null},{"url":null,"slug":"not-only-rewards-but-also-constraints","title":"Not Only Rewards But Also Constraints: Applications on Legged Robot Locomotion","date":"2023-08-24","arxiv_id":"2308.12517","repositories_listed":0,"syntology":null},{"url":null,"slug":"predator-prey-survival-pressure-is-sufficient","title":"Predator-prey survival pressure is sufficient to evolve swarming behaviors","date":"2023-08-24","arxiv_id":"2308.12624","repositories_listed":0,"syntology":null},{"url":null,"slug":"racing-towards-reinforcement-learning-based","title":"Racing Towards Reinforcement Learning based control of an Autonomous Formula SAE Car","date":"2023-08-24","arxiv_id":"2308.13088","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-informed-evolutionary","title":"Reinforcement learning informed evolutionary search for autonomous systems testing","date":"2023-08-24","arxiv_id":"2308.12762","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-based-length-controlled-generation","title":"Prompt-Based Length Controlled Generation with Reinforcement Learning","date":"2023-08-23","arxiv_id":"2308.12030","repositories_listed":0,"syntology":null},{"url":null,"slug":"boundary-rl-reinforcement-learning-for-weakly","title":"Boundary-RL: Reinforcement Learning for Weakly-Supervised Prostate Segmentation in TRUS Images","date":"2023-08-22","arxiv_id":"2308.11376","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobility-aware-computation-offloading-for","title":"Mobility-Aware Computation Offloading for Swarm Robotics using Deep Reinforcement Learning","date":"2023-08-22","arxiv_id":"2308.11154","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-opportunities-and-challenges-of-1","title":"On the Opportunities and Challenges of Offline Reinforcement Learning for Recommender Systems","date":"2023-08-22","arxiv_id":"2308.11336","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-sensor","title":"Reinforcement Learning Based Sensor Optimization for Bio-markers","date":"2023-08-21","arxiv_id":"2308.10649","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-artificial","title":"Deep Reinforcement Learning for Artificial Upwelling Energy Management","date":"2023-08-20","arxiv_id":"2308.10199","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-observer-based-reinforcement-learning","title":"An Observer-Based Reinforcement Learning Solution for Model-Following Problems","date":"2023-08-19","arxiv_id":"2308.09872","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-communication-planning-for","title":"Intelligent Communication Planning for Constrained Environmental IoT Sensing with Reinforcement Learning","date":"2023-08-19","arxiv_id":"2308.10124","repositories_listed":0,"syntology":null},{"url":null,"slug":"never-explore-repeatedly-in-multi-agent","title":"Never Explore Repeatedly in Multi-Agent Reinforcement Learning","date":"2023-08-19","arxiv_id":"2308.09909","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-robust-policy-bootstrapping-algorithm-for","title":"A Robust Policy Bootstrapping Algorithm for Multi-objective Reinforcement Learning in Non-stationary Environments","date":"2023-08-18","arxiv_id":"2308.09734","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsically-motivated-hierarchical-policy","title":"Intrinsically Motivated Hierarchical Policy Learning in Multi-objective Markov Decision Processes","date":"2023-08-18","arxiv_id":"2308.09733","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-assisted-semantic-communication-with","title":"UAV-assisted Semantic Communication with Hybrid Action Reinforcement Learning","date":"2023-08-18","arxiv_id":"2309.16713","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-decision-support-for-air-traffic","title":"Fast Decision Support for Air Traffic Management at Urban Air Mobility Vertiports using Graph Learning","date":"2023-08-17","arxiv_id":"2308.09075","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-reinforcement-learning-for-electric","title":"Federated Reinforcement Learning for Electric Vehicles Charging Control on Distribution Networks","date":"2023-08-17","arxiv_id":"2308.08792","repositories_listed":0,"syntology":null}],"record_sha256":"e85abd2f13becc81977c9bb8e3b919942d6cddba9367b036d8f35fe31ce59355","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}