{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/63","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":63,"pages_in_order":152,"rows_per_page":100,"rows":[6201,6300],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/62","next":"/task/reinforcement-learning-1/papers/64","papers":[{"url":null,"slug":"learning-robot-soccer-from-egocentric-vision","title":"Learning Robot Soccer from Egocentric Vision with Deep Reinforcement Learning","date":"2024-05-03","arxiv_id":"2405.02425","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-robust-autonomous-navigation-and","title":"Learning Robust Autonomous Navigation and Locomotion for Wheeled-Legged Robots","date":"2024-05-03","arxiv_id":"2405.01792","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-8","title":"Model-based reinforcement learning for protein backbone design","date":"2024-05-03","arxiv_id":"2405.01983","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-policy-gradient-and-actor-critic","title":"Natural Policy Gradient and Actor Critic Methods for Constrained Multi-Task Reinforcement Learning","date":"2024-05-03","arxiv_id":"2405.02456","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-sum-positional-differential-games-as-a","title":"Zero-Sum Positional Differential Games as a Framework for Robust Reinforcement Learning: Deep Q-Learning Approach","date":"2024-05-03","arxiv_id":"2405.02044","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-under","title":"Constrained Reinforcement Learning Under Model Mismatch","date":"2024-05-02","arxiv_id":"2405.01327","repositories_listed":0,"syntology":null},{"url":null,"slug":"flame-factuality-aware-alignment-for-large","title":"FLAME: Factuality-Aware Alignment for Large Language Models","date":"2024-05-02","arxiv_id":"2405.01525","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-force-control-for-legged","title":"Learning Force Control for Legged Manipulation","date":"2024-05-02","arxiv_id":"2405.01402","repositories_listed":0,"syntology":null},{"url":null,"slug":"plan-seq-learn-language-model-guided-rl-for","title":"Plan-Seq-Learn: Language Model Guided RL for Solving Long Horizon Robotics Tasks","date":"2024-05-02","arxiv_id":"2405.01534","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-edit-based-non","title":"Reinforcement Learning for Edit-Based Non-Autoregressive Neural Machine Translation","date":"2024-05-02","arxiv_id":"2405.01280","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-guided-semi-supervised","title":"Reinforcement Learning-Guided Semi-Supervised Learning","date":"2024-05-02","arxiv_id":"2405.01760","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-risk-sensitive-reinforcement-learning-1","title":"Robust Risk-Sensitive Reinforcement Learning with Conditional Value-at-Risk","date":"2024-05-02","arxiv_id":"2405.01718","repositories_listed":0,"syntology":null},{"url":null,"slug":"tabular-and-deep-reinforcement-learning-for","title":"Tabular and Deep Reinforcement Learning for Gittins Index","date":"2024-05-02","arxiv_id":"2405.01157","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigating-webai-training-agents-to-complete","title":"Navigating WebAI: Training Agents to Complete Web Tasks with Large Language Models and Reinforcement Learning","date":"2024-05-01","arxiv_id":"2405.00516","repositories_listed":0,"syntology":null},{"url":null,"slug":"queue-based-eco-driving-at-roundabouts-with","title":"Queue-based Eco-Driving at Roundabouts with Reinforcement Learning","date":"2024-05-01","arxiv_id":"2405.00625","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-sub-optimal-data-for-human-in-the","title":"Leveraging Sub-Optimal Data for Human-in-the-Loop Reinforcement Learning","date":"2024-04-30","arxiv_id":"2405.00746","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-generalist-robot-learning-from","title":"Towards Generalist Robot Learning from Internet Video: A Survey","date":"2024-04-30","arxiv_id":"2404.19664","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-policy-correction-framework-for","title":"Control Policy Correction Framework for Reinforcement Learning-based Energy Arbitrage Strategies","date":"2024-04-29","arxiv_id":"2404.18821","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-problem-solving-with","title":"Reinforcement Learning Problem Solving with Large Language Models","date":"2024-04-29","arxiv_id":"2404.18638","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-robust-multi-agent","title":"Sample-Efficient Robust Multi-Agent Reinforcement Learning in the Face of Environmental Uncertainty","date":"2024-04-29","arxiv_id":"2404.18909","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-generalizable-agents-in-text-based","title":"Towards Generalizable Agents in Text-Based Educational Environments: A Study of Integrating RL with LLMs","date":"2024-04-29","arxiv_id":"2404.18978","repositories_listed":0,"syntology":null},{"url":null,"slug":"eeg-rl-net-enhancing-eeg-mi-classification","title":"EEG_RL-Net: Enhancing EEG MI Classification through Reinforcement Learning-Optimised Graph Neural Networks","date":"2024-04-26","arxiv_id":"2405.00723","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-privacy-and-security-of-autonomous","title":"Enhancing Privacy and Security of Autonomous UAV Navigation","date":"2024-04-26","arxiv_id":"2404.17225","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalize-by-touching-tactile-ensemble-skill","title":"Generalize by Touching: Tactile Ensemble Skill Transfer for Robotic Furniture Assembly","date":"2024-04-26","arxiv_id":"2404.17684","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-transfer-for-cross-domain","title":"Knowledge Transfer for Cross-Domain Reinforcement Learning: A Systematic Review","date":"2024-04-26","arxiv_id":"2404.17687","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-8","title":"Offline Reinforcement Learning with Behavioral Supervisor Tuning","date":"2024-04-25","arxiv_id":"2404.16399","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-reinforcement-learning-for-delay","title":"Structured Reinforcement Learning for Delay-Optimal Data Transmission in Dense mmWave Networks","date":"2024-04-25","arxiv_id":"2404.16920","repositories_listed":0,"syntology":null},{"url":null,"slug":"activerir-active-audio-visual-exploration-for","title":"ActiveRIR: Active Audio-Visual Exploration for Acoustic Environment Modeling","date":"2024-04-24","arxiv_id":"2404.16216","repositories_listed":0,"syntology":null},{"url":"/paper/dpo-differential-reinforcement-learning-with","slug":"dpo-differential-reinforcement-learning-with","title":"DPO: A Differential and Pointwise Control Approach to Reinforcement Learning","date":"2024-04-24","arxiv_id":"2404.15617","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/dpo-differential-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2404.15617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15617"}},"official":null}},{"url":null,"slug":"grsn-gated-recurrent-spiking-neurons-for","title":"GRSN: Gated Recurrent Spiking Neurons for POMDPs and MARL","date":"2024-04-24","arxiv_id":"2404.15597","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-mrp-formulation-for-supervised-learning","title":"An MRP Formulation for Supervised Learning: Generalized Temporal Difference Learning Models","date":"2024-04-23","arxiv_id":"2404.15518","repositories_listed":0,"syntology":null},{"url":null,"slug":"impedance-matching-enabling-an-rl-based","title":"Impedance Matching: Enabling an RL-Based Running Jump in a Quadruped Robot","date":"2024-04-23","arxiv_id":"2404.15096","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-deep-reinforcement-learning-to-promote","title":"Using deep reinforcement learning to promote sustainable human behaviour on a common pool resource problem","date":"2024-04-23","arxiv_id":"2404.15059","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-the-edge-an-advanced-exploration-of","title":"Beyond the Edge: An Advanced Exploration of Reinforcement Learning for Mobile Edge Computing, its Applications, and Future Research Trajectories","date":"2024-04-22","arxiv_id":"2404.14238","repositories_listed":0,"syntology":null},{"url":null,"slug":"explicit-lipschitz-value-estimation-enhances","title":"Explicit Lipschitz Value Estimation Enhances Policy Robustness Against Perturbation","date":"2024-04-22","arxiv_id":"2404.13879","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairness-incentives-in-response-to-unfair","title":"Fairness Incentives in Response to Unfair Dynamic Pricing","date":"2024-04-22","arxiv_id":"2404.14620","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-off-policy-reinforcement-learning","title":"An Offline Reinforcement Learning Algorithm Customized for Multi-Task Fusion in Large-Scale Recommender Systems","date":"2024-04-19","arxiv_id":"2404.17589","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-time-risk-sensitive-reinforcement","title":"Continuous-time Risk-sensitive Reinforcement Learning via Quadratic Variation Penalty","date":"2024-04-19","arxiv_id":"2404.12598","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approach-for-1","title":"Reinforcement Learning Approach for Integrating Compressed Contexts into Knowledge Graphs","date":"2024-04-19","arxiv_id":"2404.12587","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-task-continual-offline-reinforcement","title":"Data-Incremental Continual Offline Reinforcement Learning","date":"2024-04-19","arxiv_id":"2404.12639","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-reinforcement-learning-with-1","title":"Actor-Critic Reinforcement Learning with Phased Actor","date":"2024-04-18","arxiv_id":"2404.11834","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-policy-optimization-with-temporal-logic","title":"LTL-Constrained Policy Optimization with Cycle Experience Replay","date":"2024-04-17","arxiv_id":"2404.11578","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-tour-operator-design-for-solution","title":"Learn to Tour: Operator Design For Solution Feasibility Mapping in Pickup-and-delivery Traveling Salesman Problem","date":"2024-04-17","arxiv_id":"2404.11458","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-informed-actor-critic-for","title":"Physics-informed Actor-Critic for Coordination of Virtual Inertia from Power Distribution Systems","date":"2024-04-17","arxiv_id":"2404.11149","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-optimizer-of-text-to-image-diffusion","title":"Prompt Optimizer of Text-to-Image Diffusion Models for Abstract Concept Understanding","date":"2024-04-17","arxiv_id":"2404.11589","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-discovery-of-functional-actual","title":"Automated Discovery of Functional Actual Causes in Complex Environments","date":"2024-04-16","arxiv_id":"2404.10883","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-trajectory-generalization-for-offline","title":"Offline Trajectory Generalization for Offline Reinforcement Learning","date":"2024-04-16","arxiv_id":"2404.10393","repositories_listed":0,"syntology":null},{"url":null,"slug":"settling-constant-regrets-in-linear-markov","title":"Achieving Constant Regret in Linear Markov Decision Processes","date":"2024-04-16","arxiv_id":"2404.10745","repositories_listed":0,"syntology":null},{"url":null,"slug":"simplex-decomposition-for-portfolio","title":"Simplex Decomposition for Portfolio Allocation Constraints in Reinforcement Learning","date":"2024-04-16","arxiv_id":"2404.10683","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-path-planning-for-intercostal","title":"Autonomous Path Planning for Intercostal Robotic Ultrasound Imaging Using Reinforcement Learning","date":"2024-04-15","arxiv_id":"2404.09927","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-reinforcement-learning-based-on","title":"Effective Reinforcement Learning Based on Structural Information Principles","date":"2024-04-15","arxiv_id":"2404.09760","repositories_listed":0,"syntology":null},{"url":null,"slug":"higher-replay-ratio-empowers-sample-efficient","title":"Higher Replay Ratio Empowers Sample-Efficient Multi-Agent Reinforcement Learning","date":"2024-04-15","arxiv_id":"2404.09715","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-feasibility-of-constrained-reinforcement","title":"The Feasibility of Constrained Reinforcement Learning Algorithms: A Tutorial Study","date":"2024-04-15","arxiv_id":"2404.10064","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledgeable-agents-by-offline-reinforcement","title":"Knowledgeable Agents by Offline Reinforcement Learning from Large Language Model Rollouts","date":"2024-04-14","arxiv_id":"2404.09248","repositories_listed":0,"syntology":null},{"url":null,"slug":"smartpathfinder-pushing-the-limits-of","title":"SmartPathfinder: Pushing the Limits of Heuristic Solutions for Vehicle Routing Problem with Drones Using Reinforcement Learning","date":"2024-04-13","arxiv_id":"2404.13068","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-duple-perturbation-robustness-in","title":"Efficient Duple Perturbation Robustness in Low-rank MDPs","date":"2024-04-11","arxiv_id":"2404.08089","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-policy-gradient-with-the-polyak","title":"Enhancing Policy Gradient with the Polyak Step-Size Adaption","date":"2024-04-11","arxiv_id":"2404.07525","repositories_listed":0,"syntology":null},{"url":null,"slug":"fpga-divide-and-conquer-placement-using-deep","title":"FPGA Divide-and-Conquer Placement using Deep Reinforcement Learning","date":"2024-04-11","arxiv_id":"2404.13061","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-domain-unlabeled-data-in-offline","title":"Leveraging Domain-Unlabeled Data in Offline Reinforcement Learning across Two Domains","date":"2024-04-11","arxiv_id":"2404.07465","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-sample-efficiency-of-abstractions-and","title":"On the Sample Efficiency of Abstractions and Potential-Based Reward Shaping in Reinforcement Learning","date":"2024-04-11","arxiv_id":"2404.07826","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-ensemble-kalman-filter-for-stochastic","title":"Dual Ensemble Kalman Filter for Stochastic Optimal Control","date":"2024-04-10","arxiv_id":"2404.06696","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-learning-from-suboptimal","title":"Reward Learning from Suboptimal Demonstrations with Applications in Surgical Electrocautery","date":"2024-04-10","arxiv_id":"2404.07185","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-assisted-enhanced-coverage-and-capacity","title":"UAV-Assisted Enhanced Coverage and Capacity in Dynamic MU-mMIMO IoT Systems: A Deep Reinforcement Learning Approach","date":"2024-04-10","arxiv_id":"2404.06726","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptable-recovery-behaviors-in-robotics-a","title":"Adaptable Recovery Behaviors in Robotics: A Behavior Trees and Motion Generators(BTMG) Approach for Failure Management","date":"2024-04-09","arxiv_id":"2404.06129","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-federated-reinforcement-learning","title":"Asynchronous Federated Reinforcement Learning with Policy Gradient Updates: Algorithm Design and Convergence Analysis","date":"2024-04-09","arxiv_id":"2404.08003","repositories_listed":0,"syntology":null},{"url":null,"slug":"diverse-randomized-value-functions-a-provably","title":"Diverse Randomized Value Functions: A Provably Pessimistic Approach for Offline Reinforcement Learning","date":"2024-04-09","arxiv_id":"2404.06188","repositories_listed":0,"syntology":null},{"url":null,"slug":"fgaif-aligning-large-vision-language-models","title":"FGAIF: Aligning Large Vision-Language Models with Fine-grained AI Feedback","date":"2024-04-07","arxiv_id":"2404.05046","repositories_listed":0,"syntology":null},{"url":null,"slug":"transform-then-explore-a-simple-and-effective","title":"Transform then Explore: a Simple and Effective Technique for Exploratory Combinatorial Optimization with Reinforcement Learning","date":"2024-04-06","arxiv_id":"2404.04661","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-iot-intelligence-a-transformer","title":"Enhancing IoT Intelligence: A Transformer-based Reinforcement Learning Methodology","date":"2024-04-05","arxiv_id":"2404.04205","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-reset-policy","title":"A Reinforcement Learning based Reset Policy for CDCL SAT Solvers","date":"2024-04-04","arxiv_id":"2404.03753","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributionally-robust-reinforcement-1","title":"Distributionally Robust Reinforcement Learning with Interactive Data Collection: Fundamental Hardness and Near-Optimal Algorithm","date":"2024-04-04","arxiv_id":"2404.03578","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-is-harder-than-prediction","title":"Exploration is Harder than Prediction: Cryptographically Separating Reinforcement Learning from Supervised Learning","date":"2024-04-04","arxiv_id":"2404.03774","repositories_listed":0,"syntology":null},{"url":null,"slug":"react-revealing-evolutionary-action","title":"REACT: Revealing Evolutionary Action Consequence Trajectories for Interpretable Reinforcement Learning","date":"2024-04-04","arxiv_id":"2404.03359","repositories_listed":0,"syntology":null},{"url":null,"slug":"methodology-for-interpretable-reinforcement","title":"Methodology for Interpretable Reinforcement Learning for Optimizing Mechanical Ventilation","date":"2024-04-03","arxiv_id":"2404.03105","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-categorical","title":"Reinforcement Learning in Categorical Cybernetics","date":"2024-04-03","arxiv_id":"2404.02688","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-exploration-in-bayesian-model-based","title":"Active Exploration in Bayesian Model-based Reinforcement Learning for Robot Manipulation","date":"2024-04-02","arxiv_id":"2404.01867","repositories_listed":0,"syntology":null},{"url":null,"slug":"asymptotics-of-language-model-alignment","title":"Asymptotics of Language Model Alignment","date":"2024-04-02","arxiv_id":"2404.01730","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergence-of-chemotactic-strategies-with","title":"Emergence of Chemotactic Strategies with Multi-Agent Reinforcement Learning","date":"2024-04-02","arxiv_id":"2404.01999","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-exploration-all-you-need-effective","title":"Is Exploration All You Need? Effective Exploration Characteristics for Transfer in Reinforcement Learning","date":"2024-04-02","arxiv_id":"2404.02235","repositories_listed":0,"syntology":null},{"url":null,"slug":"mtlight-efficient-multi-task-reinforcement","title":"MTLight: Efficient Multi-Task Reinforcement Learning for Traffic Signal Control","date":"2024-04-01","arxiv_id":"2404.00886","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-off-policy-with-model-based","title":"Learning Off-policy with Model-based Intrinsic Motivation For Active Online Exploration","date":"2024-03-31","arxiv_id":"2404.00651","repositories_listed":0,"syntology":null},{"url":null,"slug":"utilizing-maximum-mean-discrepancy-barycenter","title":"Utilizing Maximum Mean Discrepancy Barycenter for Propagating the Uncertainty of Value Functions in Reinforcement Learning","date":"2024-03-31","arxiv_id":"2404.00686","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-on-large-language-model-enhanced","title":"Survey on Large Language Model-Enhanced Reinforcement Learning: Concept, Taxonomy, and Methods","date":"2024-03-30","arxiv_id":"2404.00282","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctrl-sim-reactive-and-controllable-driving","title":"CtRL-Sim: Reactive and Controllable Driving Agents with Offline Reinforcement Learning","date":"2024-03-29","arxiv_id":"2403.19918","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-visual-quadrupedal-loco-manipulation","title":"Learning Visual Quadrupedal Loco-Manipulation from Demonstrations","date":"2024-03-29","arxiv_id":"2403.20328","repositories_listed":0,"syntology":null},{"url":null,"slug":"molecular-generative-adversarial-network-with","title":"Molecular Generative Adversarial Network with Multi-Property Optimization","date":"2024-03-29","arxiv_id":"2404.00081","repositories_listed":0,"syntology":null},{"url":null,"slug":"nonparametric-bellman-mappings-for","title":"Nonparametric Bellman Mappings for Reinforcement Learning: Application to Robust Adaptive Filtering","date":"2024-03-29","arxiv_id":"2403.20020","repositories_listed":0,"syntology":null},{"url":null,"slug":"jointly-training-and-pruning-cnns-via","title":"Jointly Training and Pruning CNNs via Learnable Agent Guidance and Alignment","date":"2024-03-28","arxiv_id":"2403.19490","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-imitation-learning-from-multiple","title":"Offline Imitation Learning from Multiple Baselines with Applications to Compiler Optimization","date":"2024-03-28","arxiv_id":"2403.19462","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-agent-based-market","title":"Reinforcement Learning in Agent-Based Market Simulation: Unveiling Realistic Stylized Facts and Behavior","date":"2024-03-28","arxiv_id":"2403.19781","repositories_listed":0,"syntology":null},{"url":null,"slug":"cat-constraints-as-terminations-for-legged","title":"CaT: Constraints as Terminations for Legged Locomotion Reinforcement Learning","date":"2024-03-27","arxiv_id":"2403.18765","repositories_listed":0,"syntology":null},{"url":null,"slug":"fpga-based-neural-thrust-controller-for-uavs","title":"FPGA-Based Neural Thrust Controller for UAVs","date":"2024-03-27","arxiv_id":"2403.18703","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-deraining-via-self-supervised","title":"Image Deraining via Self-supervised Reinforcement Learning","date":"2024-03-27","arxiv_id":"2403.18270","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-and-short-term-constraints-driven-safe","title":"Long and Short-Term Constraints Driven Safe Reinforcement Learning for Autonomous Driving","date":"2024-03-27","arxiv_id":"2403.18209","repositories_listed":0,"syntology":null},{"url":null,"slug":"lord-large-models-based-opposite-reward","title":"LORD: Large Models based Opposite Reward Design for Autonomous Driving","date":"2024-03-27","arxiv_id":"2403.18965","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-model-checking-of-stochastic","title":"Probabilistic Model Checking of Stochastic Reinforcement Learning Policies","date":"2024-03-27","arxiv_id":"2403.18725","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-and-visual-explanation-for-black","title":"Robustness and Visual Explanation for Black Box Image, Video, and ECG Signal Classification with Reinforcement Learning","date":"2024-03-27","arxiv_id":"2403.18985","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-and-robust-reinforcement-learning","title":"Safe and Robust Reinforcement Learning: Principles and Practice","date":"2024-03-27","arxiv_id":"2403.18539","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-human-centered-construction-robotics","title":"Towards Human-Centered Construction Robotics: A Reinforcement Learning-Driven Companion Robot for Contextually Assisting Carpentry Workers","date":"2024-03-27","arxiv_id":"2403.19060","repositories_listed":0,"syntology":null},{"url":null,"slug":"depending-on-yourself-when-you-should","title":"Depending on yourself when you should: Mentoring LLM with RL agents to become the master in cybersecurity games","date":"2024-03-26","arxiv_id":"2403.17674","repositories_listed":0,"syntology":null}],"record_sha256":"3ba8449a57f86e5aa41e7cb66b1272c4dca4757b6c04e1032ddf160ff3cbaae2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}