{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/55","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":55,"pages_in_order":135,"rows_per_page":100,"rows":[5401,5500],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/54","next":"/task/reinforcement-learning-2/papers/56","papers":[{"url":null,"slug":"improving-offline-reinforcement-learning-with","title":"Improving Offline Reinforcement Learning with Inaccurate Simulators","date":"2024-05-07","arxiv_id":"2405.04307","repositories_listed":0,"syntology":null},{"url":null,"slug":"racer-epistemic-risk-sensitive-rl-enables","title":"RACER: Epistemic Risk-Sensitive RL Enables Fast Driving with Fewer Crashes","date":"2024-05-07","arxiv_id":"2405.04714","repositories_listed":0,"syntology":null},{"url":null,"slug":"torchdriveenv-a-reinforcement-learning","title":"TorchDriveEnv: A Reinforcement Learning Benchmark for Autonomous Driving with Reactive, Realistic, and Diverse Non-Playable Characters","date":"2024-05-07","arxiv_id":"2405.04491","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-reinforcement-learning-with-1","title":"Federated Reinforcement Learning with Constraint Heterogeneity","date":"2024-05-06","arxiv_id":"2405.03236","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-convergence-and-sample-complexity-1","title":"Finite-Time Convergence and Sample Complexity of Actor-Critic Multi-Objective Reinforcement Learning","date":"2024-05-05","arxiv_id":"2405.03082","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-learned-non","title":"Safe Reinforcement Learning with Learned Non-Markovian Safety Constraints","date":"2024-05-05","arxiv_id":"2405.03005","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-safe-set-algorithm-for-provably-safe","title":"Implicit Safe Set Algorithm for Provably Safe Reinforcement Learning","date":"2024-05-04","arxiv_id":"2405.02754","repositories_listed":0,"syntology":null},{"url":null,"slug":"taming-equilibrium-bias-in-risk-sensitive","title":"Taming Equilibrium Bias in Risk-Sensitive Multi-Agent Reinforcement Learning","date":"2024-05-04","arxiv_id":"2405.02724","repositories_listed":0,"syntology":null},{"url":null,"slug":"uduc-an-uncertainty-driven-approach-for","title":"UDUC: An Uncertainty-driven Approach for Learning-based Robust Control","date":"2024-05-04","arxiv_id":"2405.02598","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-robot-soccer-from-egocentric-vision","title":"Learning Robot Soccer from Egocentric Vision with Deep Reinforcement Learning","date":"2024-05-03","arxiv_id":"2405.02425","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-8","title":"Model-based reinforcement learning for protein backbone design","date":"2024-05-03","arxiv_id":"2405.01983","repositories_listed":0,"syntology":null},{"url":null,"slug":"socialgfs-learning-social-gradient-fields-for","title":"SocialGFs: Learning Social Gradient Fields for Multi-Agent Reinforcement Learning","date":"2024-05-03","arxiv_id":"2405.01839","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-sum-positional-differential-games-as-a","title":"Zero-Sum Positional Differential Games as a Framework for Robust Reinforcement Learning: Deep Q-Learning Approach","date":"2024-05-03","arxiv_id":"2405.02044","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-imitation-for-manipulator-control","title":"Behavior Imitation for Manipulator Control and Grasping with Deep Reinforcement Learning","date":"2024-05-02","arxiv_id":"2405.01284","repositories_listed":0,"syntology":null},{"url":null,"slug":"citylearn-v2-energy-flexible-resilient","title":"CityLearn v2: Energy-flexible, resilient, occupant-centric, and carbon-aware management of grid-interactive communities","date":"2024-05-02","arxiv_id":"2405.03848","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-under","title":"Constrained Reinforcement Learning Under Model Mismatch","date":"2024-05-02","arxiv_id":"2405.01327","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-conditioned-reinforcement-learning-for","title":"Goal-conditioned reinforcement learning for ultrasound navigation guidance","date":"2024-05-02","arxiv_id":"2405.01409","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-hybrid-resource-allocation-in-mec","title":"Intelligent Hybrid Resource Allocation in MEC-assisted RAN Slicing Network","date":"2024-05-02","arxiv_id":"2405.17436","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-edit-based-non","title":"Reinforcement Learning for Edit-Based Non-Autoregressive Neural Machine Translation","date":"2024-05-02","arxiv_id":"2405.01280","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-guided-semi-supervised","title":"Reinforcement Learning-Guided Semi-Supervised Learning","date":"2024-05-02","arxiv_id":"2405.01760","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-risk-sensitive-reinforcement-learning-1","title":"Robust Risk-Sensitive Reinforcement Learning with Conditional Value-at-Risk","date":"2024-05-02","arxiv_id":"2405.01718","repositories_listed":0,"syntology":null},{"url":null,"slug":"tabular-and-deep-reinforcement-learning-for","title":"Tabular and Deep Reinforcement Learning for Gittins Index","date":"2024-05-02","arxiv_id":"2405.01157","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-interpretable-reinforcement-learning-1","title":"Towards Interpretable Reinforcement Learning with Constrained Normalizing Flow Policies","date":"2024-05-02","arxiv_id":"2405.01198","repositories_listed":0,"syntology":null},{"url":null,"slug":"employing-federated-learning-for-training","title":"Employing Federated Learning for Training Autonomous HVAC Systems","date":"2024-05-01","arxiv_id":"2405.00389","repositories_listed":0,"syntology":null},{"url":null,"slug":"mf-oml-online-mean-field-reinforcement","title":"MF-OML: Online Mean-Field Reinforcement Learning with Occupation Measures for Large Population Games","date":"2024-05-01","arxiv_id":"2405.00282","repositories_listed":0,"syntology":null},{"url":null,"slug":"portfolio-management-using-deep-reinforcement","title":"Portfolio Management using Deep Reinforcement Learning","date":"2024-05-01","arxiv_id":"2405.01604","repositories_listed":0,"syntology":null},{"url":null,"slug":"queue-based-eco-driving-at-roundabouts-with","title":"Queue-based Eco-Driving at Roundabouts with Reinforcement Learning","date":"2024-05-01","arxiv_id":"2405.00625","repositories_listed":0,"syntology":null},{"url":null,"slug":"bias-mitigation-via-compensation-a","title":"AI, Pluralism, and (Social) Compensation","date":"2024-04-30","arxiv_id":"2404.19256","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-policy-correction-framework-for","title":"Control Policy Correction Framework for Reinforcement Learning-based Energy Arbitrage Strategies","date":"2024-04-29","arxiv_id":"2404.18821","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-problem-solving-with","title":"Reinforcement Learning Problem Solving with Large Language Models","date":"2024-04-29","arxiv_id":"2404.18638","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-rational-reinforcement-learning-and","title":"Resource-rational reinforcement learning and sensorimotor causal states, and resource-rational maximiners","date":"2024-04-29","arxiv_id":"2404.18775","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-training-superconducting-neuromorphic","title":"Self-training superconducting neuromorphic circuits using reinforcement learning rules","date":"2024-04-29","arxiv_id":"2404.18774","repositories_listed":0,"syntology":null},{"url":null,"slug":"verco-learning-coordinated-verbal","title":"Verco: Learning Coordinated Verbal Communication for Multi-agent Reinforcement Learning","date":"2024-04-27","arxiv_id":"2404.17780","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-transfer-for-cross-domain","title":"Knowledge Transfer for Cross-Domain Reinforcement Learning: A Systematic Review","date":"2024-04-26","arxiv_id":"2404.17687","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-8","title":"Offline Reinforcement Learning with Behavioral Supervisor Tuning","date":"2024-04-25","arxiv_id":"2404.16399","repositories_listed":0,"syntology":null},{"url":"/paper/dpo-differential-reinforcement-learning-with","slug":"dpo-differential-reinforcement-learning-with","title":"DPO: A Differential and Pointwise Control Approach to Reinforcement Learning","date":"2024-04-24","arxiv_id":"2404.15617","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/dpo-differential-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2404.15617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15617"}},"official":null}},{"url":null,"slug":"grsn-gated-recurrent-spiking-neurons-for","title":"GRSN: Gated Recurrent Spiking Neurons for POMDPs and MARL","date":"2024-04-24","arxiv_id":"2404.15597","repositories_listed":0,"syntology":null},{"url":null,"slug":"cache-aware-reinforcement-learning-in-large","title":"Cache-Aware Reinforcement Learning in Large-Scale Recommender Systems","date":"2024-04-23","arxiv_id":"2404.14961","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-high-speed-cruising-performance-of","title":"Enhancing High-Speed Cruising Performance of Autonomous Vehicles through Integrated Deep Reinforcement Learning Framework","date":"2024-04-23","arxiv_id":"2404.14713","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-reinforcement-learning-via-1","title":"Evolutionary Reinforcement Learning via Cooperative Coevolution","date":"2024-04-23","arxiv_id":"2404.14763","repositories_listed":0,"syntology":null},{"url":null,"slug":"multistop-solving-functional-equations-with","title":"MultiSTOP: Solving Functional Equations with Reinforcement Learning","date":"2024-04-23","arxiv_id":"2404.14909","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-power-of-resets-in-online-reinforcement","title":"The Power of Resets in Online Reinforcement Learning","date":"2024-04-23","arxiv_id":"2404.15417","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-deep-reinforcement-learning-to-promote","title":"Using deep reinforcement learning to promote sustainable human behaviour on a common pool resource problem","date":"2024-04-23","arxiv_id":"2404.15059","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-black-box-model-inversion","title":"Distributional Black-Box Model Inversion Attack with Multi-Agent Reinforcement Learning","date":"2024-04-22","arxiv_id":"2404.13860","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-control-barrier-functions-and-their","title":"Learning Control Barrier Functions and their application in Reinforcement Learning: A Survey","date":"2024-04-22","arxiv_id":"2404.16879","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-stability-of-lipschitz-continuous","title":"On the stability of Lipschitz continuous control problems and its application to reinforcement learning","date":"2024-04-20","arxiv_id":"2404.13316","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-time-risk-sensitive-reinforcement","title":"Continuous-time Risk-sensitive Reinforcement Learning via Quadratic Variation Penalty","date":"2024-04-19","arxiv_id":"2404.12598","repositories_listed":0,"syntology":null},{"url":null,"slug":"mapping-social-choice-theory-to-rlhf","title":"Mapping Social Choice Theory to RLHF","date":"2024-04-19","arxiv_id":"2404.13038","repositories_listed":0,"syntology":null},{"url":null,"slug":"mm-phyrlhf-reinforcement-learning-framework","title":"MM-PhyRLHF: Reinforcement Learning Framework for Multimodal Physics Question-Answering","date":"2024-04-19","arxiv_id":"2404.12926","repositories_listed":0,"syntology":null},{"url":null,"slug":"random-network-distillation-based-deep","title":"Random Network Distillation Based Deep Reinforcement Learning for AGV Path Planning","date":"2024-04-19","arxiv_id":"2404.12594","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approach-for-1","title":"Reinforcement Learning Approach for Integrating Compressed Contexts into Knowledge Graphs","date":"2024-04-19","arxiv_id":"2404.12587","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-task-continual-offline-reinforcement","title":"Data-Incremental Continual Offline Reinforcement Learning","date":"2024-04-19","arxiv_id":"2404.12639","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-stitching-in-reinforcement-learning","title":"Zero-Shot Stitching in Reinforcement Learning using Relative Representations","date":"2024-04-19","arxiv_id":"2404.12917","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-reinforcement-learning-with-1","title":"Actor-Critic Reinforcement Learning with Phased Actor","date":"2024-04-18","arxiv_id":"2404.11834","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-policy-optimization-with-temporal-logic","title":"LTL-Constrained Policy Optimization with Cycle Experience Replay","date":"2024-04-17","arxiv_id":"2404.11578","repositories_listed":0,"syntology":null},{"url":null,"slug":"function-approximation-for-reinforcement","title":"Function Approximation for Reinforcement Learning Controller for Energy from Spread Waves","date":"2024-04-17","arxiv_id":"2404.10991","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-re-calibration-of-quantum-devices","title":"Automatic re-calibration of quantum devices by reinforcement learning","date":"2024-04-16","arxiv_id":"2404.10726","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-control-reinforcement-learning","title":"Continuous Control Reinforcement Learning: Distributed Distributional DrQ Algorithms","date":"2024-04-16","arxiv_id":"2404.10645","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-trajectory-generalization-for-offline","title":"Offline Trajectory Generalization for Offline Reinforcement Learning","date":"2024-04-16","arxiv_id":"2404.10393","repositories_listed":0,"syntology":null},{"url":"/paper/randomized-exploration-in-cooperative-multi","slug":"randomized-exploration-in-cooperative-multi","title":"Randomized Exploration in Cooperative Multi-Agent Reinforcement Learning","date":"2024-04-16","arxiv_id":"2404.10728","repositories_listed":0,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/randomized-exploration-in-cooperative-multi#ran","syntology_url":"https://syntology.ai/paper/2404.10728","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10728"}},"official":null}},{"url":null,"slug":"simplex-decomposition-for-portfolio","title":"Simplex Decomposition for Portfolio Allocation Constraints in Reinforcement Learning","date":"2024-04-16","arxiv_id":"2404.10683","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-research-community-in-interpretable","title":"Towards a Research Community in Interpretable Reinforcement Learning: the InterpPol Workshop","date":"2024-04-16","arxiv_id":"2404.10906","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-path-planning-for-intercostal","title":"Autonomous Path Planning for Intercostal Robotic Ultrasound Imaging Using Reinforcement Learning","date":"2024-04-15","arxiv_id":"2404.09927","repositories_listed":0,"syntology":null},{"url":null,"slug":"deceiving-to-enlighten-coaxing-llms-to-self","title":"Reinforcement Learning from Multi-role Debates as Feedback for Bias Mitigation in LLMs","date":"2024-04-15","arxiv_id":"2404.10160","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-reinforcement-learning-based-on","title":"Effective Reinforcement Learning Based on Structural Information Principles","date":"2024-04-15","arxiv_id":"2404.09760","repositories_listed":0,"syntology":null},{"url":null,"slug":"eyeformer-predicting-personalized-scanpaths","title":"EyeFormer: Predicting Personalized Scanpaths with Transformer-Guided Reinforcement Learning","date":"2024-04-15","arxiv_id":"2404.10163","repositories_listed":0,"syntology":null},{"url":null,"slug":"higher-replay-ratio-empowers-sample-efficient","title":"Higher Replay Ratio Empowers Sample-Efficient Multi-Agent Reinforcement Learning","date":"2024-04-15","arxiv_id":"2404.09715","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effects-of-fine-tuning-language-models","title":"On the Effects of Fine-tuning Language Models for Text-Based Reinforcement Learning","date":"2024-04-15","arxiv_id":"2404.10174","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-feasibility-of-constrained-reinforcement","title":"The Feasibility of Constrained Reinforcement Learning Algorithms: A Tutorial Study","date":"2024-04-15","arxiv_id":"2404.10064","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledgeable-agents-by-offline-reinforcement","title":"Knowledgeable Agents by Offline Reinforcement Learning from Large Language Model Rollouts","date":"2024-04-14","arxiv_id":"2404.09248","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-offline-quantum-reinforcement","title":"Model-based Offline Quantum Reinforcement Learning","date":"2024-04-14","arxiv_id":"2404.10017","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-learning-for-control-oriented","title":"Active Learning for Control-Oriented Identification of Nonlinear Systems","date":"2024-04-13","arxiv_id":"2404.09030","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-on-the-constraint","title":"Safe Reinforcement Learning on the Constraint Manifold: Theory and Applications","date":"2024-04-13","arxiv_id":"2404.09080","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-forest-fire-prevention-deep","title":"Advancing Forest Fire Prevention: Deep Reinforcement Learning for Effective Firebreak Placement","date":"2024-04-12","arxiv_id":"2404.08523","repositories_listed":0,"syntology":null},{"url":null,"slug":"agile-and-versatile-bipedal-robot-tracking","title":"Agile and versatile bipedal robot tracking control through reinforcement learning","date":"2024-04-12","arxiv_id":"2404.08246","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlemmo-evolutionary-multimodal-optimization","title":"RLEMMO: Evolutionary Multimodal Optimization Assisted By Deep Reinforcement Learning","date":"2024-04-12","arxiv_id":"2404.08242","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlhf-deciphered-a-critical-analysis-of","title":"RLHF Deciphered: A Critical Analysis of Reinforcement Learning from Human Feedback for LLMs","date":"2024-04-12","arxiv_id":"2404.08555","repositories_listed":0,"syntology":null},{"url":null,"slug":"sir-rl-reinforcement-learning-for-optimized","title":"SIR-RL: Reinforcement Learning for Optimized Policy Control during Epidemiological Outbreaks in Emerging Market and Developing Economies","date":"2024-04-12","arxiv_id":"2404.08423","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentially-private-reinforcement-learning-1","title":"Differentially Private Reinforcement Learning with Self-Play","date":"2024-04-11","arxiv_id":"2404.07559","repositories_listed":0,"syntology":null},{"url":null,"slug":"fpga-divide-and-conquer-placement-using-deep","title":"FPGA Divide-and-Conquer Placement using Deep Reinforcement Learning","date":"2024-04-11","arxiv_id":"2404.13061","repositories_listed":0,"syntology":null},{"url":null,"slug":"r2-indicator-and-deep-reinforcement-learning","title":"R2 Indicator and Deep Reinforcement Learning Enhanced Adaptive Multi-Objective Evolutionary Algorithm","date":"2024-04-11","arxiv_id":"2404.08161","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-enabled-collaborative-beamforming-via","title":"UAV-enabled Collaborative Beamforming via Multi-Agent Deep Reinforcement Learning","date":"2024-04-11","arxiv_id":"2404.07453","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-ensemble-kalman-filter-for-stochastic","title":"Dual Ensemble Kalman Filter for Stochastic Optimal Control","date":"2024-04-10","arxiv_id":"2404.06696","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-reinforcement-learning-for-media","title":"Structured Reinforcement Learning for Media Streaming at the Wireless Edge","date":"2024-04-10","arxiv_id":"2404.07315","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-approach","title":"Deep Reinforcement Learning-Based Approach for a Single Vehicle Persistent Surveillance Problem with Fuel Constraints","date":"2024-04-09","arxiv_id":"2404.06423","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-multi-task-reinforcement-learning-1","title":"Efficient Multi-Task Reinforcement Learning via Task-Specific Action Correction","date":"2024-04-09","arxiv_id":"2404.05950","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-pre-trained-transformer-for-2","title":"Generative Pre-Trained Transformer for Symbolic Regression Base In-Context Reinforcement Learning","date":"2024-04-09","arxiv_id":"2404.06330","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-reinforcement-learning-for","title":"Graph Reinforcement Learning for Combinatorial Optimization: A Survey and Unifying Perspective","date":"2024-04-09","arxiv_id":"2404.06492","repositories_listed":0,"syntology":null},{"url":null,"slug":"chiplet-placement-order-exploration-based-on","title":"Chiplet Placement Order Exploration Based on Learning to Rank with Graph Representation","date":"2024-04-07","arxiv_id":"2404.04943","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-control-for","title":"Deep Reinforcement Learning Control for Disturbance Rejection in a Nonlinear Dynamic System with Parametric Uncertainty","date":"2024-04-06","arxiv_id":"2404.04699","repositories_listed":0,"syntology":null},{"url":null,"slug":"structurally-flexible-neural-networks","title":"Structurally Flexible Neural Networks: Evolving the Building Blocks for General Agents","date":"2024-04-06","arxiv_id":"2404.15193","repositories_listed":0,"syntology":null},{"url":null,"slug":"demonstration-guided-multi-objective","title":"Demonstration Guided Multi-Objective Reinforcement Learning","date":"2024-04-05","arxiv_id":"2404.03997","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-iot-intelligence-a-transformer","title":"Enhancing IoT Intelligence: A Transformer-based Reinforcement Learning Methodology","date":"2024-04-05","arxiv_id":"2404.04205","repositories_listed":0,"syntology":null},{"url":null,"slug":"heterogeneous-multi-agent-reinforcement-2","title":"Heterogeneous Multi-Agent Reinforcement Learning for Zero-Shot Scalable Collaboration","date":"2024-04-05","arxiv_id":"2404.03869","repositories_listed":0,"syntology":null},{"url":null,"slug":"intervention-assisted-policy-gradient-methods","title":"Intervention-Assisted Policy Gradient Methods for Online Stochastic Queuing Network Optimization: Technical Report","date":"2024-04-05","arxiv_id":"2404.04106","repositories_listed":0,"syntology":null},{"url":null,"slug":"pixel-wise-rl-on-diffusion-models","title":"Pixel-wise RL on Diffusion Models: Reinforcement Learning from Rich Feedback","date":"2024-04-05","arxiv_id":"2404.04356","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-reset-policy","title":"A Reinforcement Learning based Reset Policy for CDCL SAT Solvers","date":"2024-04-04","arxiv_id":"2404.03753","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-small-language-models-help-large-language","title":"Can Small Language Models Help Large Language Models Reason Better?: LM-Guided Chain-of-Thought","date":"2024-04-04","arxiv_id":"2404.03414","repositories_listed":0,"syntology":null},{"url":null,"slug":"react-revealing-evolutionary-action","title":"REACT: Revealing Evolutionary Action Consequence Trajectories for Interpretable Reinforcement Learning","date":"2024-04-04","arxiv_id":"2404.03359","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-organized-arrival-system-for-urban-air","title":"Self-organized free-flight arrival for urban air mobility","date":"2024-04-04","arxiv_id":"2404.03710","repositories_listed":0,"syntology":null}],"record_sha256":"a0d602fc8c1768d3255f6271b5f897a68ef5a023c3834f3464a1df2d2da827ae","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}