{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/60","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":60,"pages_in_order":152,"rows_per_page":100,"rows":[5901,6000],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/59","next":"/task/reinforcement-learning-1/papers/61","papers":[{"url":null,"slug":"benchmarking-reinforcement-learning-methods","title":"Benchmarking Reinforcement Learning Methods for Dexterous Robotic Manipulation with a Three-Fingered Gripper","date":"2024-08-27","arxiv_id":"2408.14747","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-impact-of-multiple-der","title":"Evaluating the Impact of Multiple DER Aggregators on Wholesale Energy Markets: A Hybrid Mean Field Approach","date":"2024-08-27","arxiv_id":"2409.00107","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimization-solution-functions-as","title":"Optimization Solution Functions as Deterministic Policies for Offline Reinforcement Learning","date":"2024-08-27","arxiv_id":"2408.15368","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-training-of-first-and-second","title":"Simultaneous Training of First- and Second-Order Optimizers in Population-Based Reinforcement Learning","date":"2024-08-27","arxiv_id":"2408.15421","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-to-online-reinforcement-learning","title":"Unsupervised-to-Online Reinforcement Learning","date":"2024-08-27","arxiv_id":"2408.14785","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamicroutegpt-a-real-time-multi-vehicle","title":"DynamicRouteGPT: A Real-Time Multi-Vehicle Dynamic Navigation Framework Based on Large Language Models","date":"2024-08-26","arxiv_id":"2408.14185","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-9","title":"Model-Based Reinforcement Learning for Control of Strongly-Disturbed Unsteady Aerodynamic Flows","date":"2024-08-26","arxiv_id":"2408.14685","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-target-assignment-and-path","title":"Multi-Agent Target Assignment and Path Finding for Intelligent Warehouse: A Cooperative Multi-Agent Deep Reinforcement Learning Perspective","date":"2024-08-25","arxiv_id":"2408.13750","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-for-continual-rl-via","title":"Data Augmentation for Continual RL via Adversarial Gradient Episodic Memory","date":"2024-08-24","arxiv_id":"2408.13452","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforced-dag-learning-without","title":"Reinforcement Learning for Causal Discovery without Acyclicity Constraints","date":"2024-08-24","arxiv_id":"2408.13448","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-state-disentanglement-in-causal","title":"Rethinking State Disentanglement in Causal Reinforcement Learning","date":"2024-08-24","arxiv_id":"2408.13498","repositories_listed":0,"syntology":null},{"url":null,"slug":"localized-observation-abstraction-using","title":"Localized Observation Abstraction Using Piecewise Linear Spatial Decay for Reinforcement Learning in Combat Simulations","date":"2024-08-23","arxiv_id":"2408.13328","repositories_listed":0,"syntology":null},{"url":null,"slug":"mastering-the-digital-art-of-war-developing","title":"Mastering the Digital Art of War: Developing Intelligent Combat Simulation Agents for Wargaming Using Hierarchical Reinforcement Learning","date":"2024-08-23","arxiv_id":"2408.13333","repositories_listed":0,"syntology":null},{"url":null,"slug":"sumo-search-based-uncertainty-estimation-for","title":"SUMO: Search-Based Uncertainty Estimation for Model-Based Offline Reinforcement Learning","date":"2024-08-23","arxiv_id":"2408.12970","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-act-prioritization-strategies-for","title":"Balancing Act: Prioritization Strategies for LLM-Designed Restless Bandit Rewards","date":"2024-08-22","arxiv_id":"2408.12112","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adaptation-for-offline-reinforcement","title":"Domain Adaptation for Offline Reinforcement Learning with Limited Samples","date":"2024-08-22","arxiv_id":"2408.12136","repositories_listed":0,"syntology":null},{"url":null,"slug":"query-efficient-video-adversarial-attack-with","title":"Query-Efficient Video Adversarial Attack with Stylized Logo","date":"2024-08-22","arxiv_id":"2408.12099","repositories_listed":0,"syntology":null},{"url":null,"slug":"advances-in-preference-based-reinforcement","title":"Advances in Preference-based Reinforcement Learning: A Review","date":"2024-08-21","arxiv_id":"2408.11943","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-part-based-representations-for","title":"Using Part-based Representations for Explainable Deep Reinforcement Learning","date":"2024-08-21","arxiv_id":"2408.11455","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-multi-modal-input-token-mixer","title":"Integrating Multi-Modal Input Token Mixer Into Mamba-Based Decision Models: Decision MetaMamba","date":"2024-08-20","arxiv_id":"2408.10517","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-model-based-reinforcement-learning","title":"Offline Model-Based Reinforcement Learning with Anti-Exploration","date":"2024-08-20","arxiv_id":"2408.10713","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-evolution-of-reinforcement-learning-in","title":"The Evolution of Reinforcement Learning in Quantitative Finance: A Survey","date":"2024-08-20","arxiv_id":"2408.10932","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-exploration-in-deep-reinforcement","title":"Efficient Exploration in Deep Reinforcement Learning: A Novel Bayesian Actor-Critic Algorithm","date":"2024-08-19","arxiv_id":"2408.10055","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-reinforcement-learning-through","title":"Enhancing Reinforcement Learning Through Guided Search","date":"2024-08-19","arxiv_id":"2408.10113","repositories_listed":0,"syntology":null},{"url":null,"slug":"mallight-influence-aware-coordinated-traffic","title":"MalLight: Influence-Aware Coordinated Traffic Signal Control for Traffic Signal Malfunctions","date":"2024-08-19","arxiv_id":"2408.09768","repositories_listed":0,"syntology":null},{"url":null,"slug":"world-models-increase-autonomy-in","title":"World Models Increase Autonomy in Reinforcement Learning","date":"2024-08-19","arxiv_id":"2408.09807","repositories_listed":0,"syntology":null},{"url":null,"slug":"ancestral-reinforcement-learning-unifying","title":"Ancestral Reinforcement Learning: Unifying Zeroth-Order Optimization and Genetic Algorithms for Reinforcement Learning","date":"2024-08-18","arxiv_id":"2408.09493","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-adaptive-speed","title":"Reinforcement learning-based adaptive speed controllers in mixed autonomy condition","date":"2024-08-17","arxiv_id":"2408.09145","repositories_listed":0,"syntology":null},{"url":null,"slug":"cat-caution-aware-transfer-in-reinforcement","title":"CAT: Caution Aware Transfer in Reinforcement Learning via Distributional Risk","date":"2024-08-16","arxiv_id":"2408.08812","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-rl-as-a-minimalist-approach-to","title":"Model-based RL as a Minimalist Approach to Horizon-Free and Second-Order Bounds","date":"2024-08-16","arxiv_id":"2408.08994","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-user-journeys-in-pharma-e-commerce","title":"Adaptive User Journeys in Pharma E-Commerce with Reinforcement Learning: Insights from SwipeRx","date":"2024-08-15","arxiv_id":"2408.08024","repositories_listed":0,"syntology":null},{"url":null,"slug":"ireca-intrinsic-reward-enhanced-context-aware","title":"BCR-DRL: Behavior- and Context-aware Reward for Deep Reinforcement Learning in Human-AI Coordination","date":"2024-08-15","arxiv_id":"2408.07877","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-behavior-modification-for-expressive","title":"Online Behavior Modification for Expressive User Control of RL-Trained Robots","date":"2024-08-15","arxiv_id":"2408.16776","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-prompting-with-episodic","title":"Large Language Models Prompting With Episodic Memory","date":"2024-08-14","arxiv_id":"2408.07465","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-reinforcement-learning-with-high","title":"Off-Policy Reinforcement Learning with High Dimensional Reward","date":"2024-08-14","arxiv_id":"2408.07660","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-of-information-and-reward-specification","title":"Value of Information and Reward Specification in Active Inference and POMDPs","date":"2024-08-13","arxiv_id":"2408.06542","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-optimization-of-curriculum-learning","title":"Online Optimization of Curriculum Learning Schedules using Evolutionary Optimization","date":"2024-08-12","arxiv_id":"2408.06068","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-bandit-whisperer-communication-learning","title":"The Bandit Whisperer: Communication Learning for Restless Bandits","date":"2024-08-11","arxiv_id":"2408.05686","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-portfolio-with-two-sided","title":"Optimizing Portfolio with Two-Sided Transactions and Lending: A Reinforcement Learning Framework","date":"2024-08-09","arxiv_id":"2408.05382","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-the-design-of","title":"Deep Reinforcement Learning for the Design of Metamaterial Mechanisms with Functional Compliance Control","date":"2024-08-08","arxiv_id":"2408.04376","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-reinforcement-learning-breaks-sample","title":"Hybrid Reinforcement Learning Breaks Sample Size Barriers in Linear MDPs","date":"2024-08-08","arxiv_id":"2408.04526","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-robotics-a","title":"Deep Reinforcement Learning for Robotics: A Survey of Real-World Successes","date":"2024-08-07","arxiv_id":"2408.03539","repositories_listed":0,"syntology":null},{"url":null,"slug":"navinact-combining-navigation-and-imitation","title":"PLANRL: A Motion Planning and Imitation Learning Framework to Bootstrap Reinforcement Learning","date":"2024-08-07","arxiv_id":"2408.04054","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-03018","title":"Integrating Controllable Motion Skills from Demonstrations","date":"2024-08-06","arxiv_id":"2408.03018","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-03166","title":"CADRL: Category-aware Dual-agent Reinforcement Learning for Explainable Recommendations over Knowledge Graphs","date":"2024-08-06","arxiv_id":"2408.03166","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-optimal-controller-for-discrete","title":"Model-free optimal controller for discrete-time Markovian jump linear systems: A Q-learning approach","date":"2024-08-06","arxiv_id":"2408.03077","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02349","title":"Active Sensing of Knee Osteoarthritis Progression with Reinforcement Learning","date":"2024-08-05","arxiv_id":"2408.02349","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02489","title":"Full error analysis of policy gradient learning algorithms for exploratory linear quadratic mean-field control problem in continuous time with common noise","date":"2024-08-05","arxiv_id":"2408.02489","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01999","title":"Reinforcement Learning for an Efficient and Effective Malware Investigation during Cyber Incident Response","date":"2024-08-04","arxiv_id":"2408.01999","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01072","title":"A Survey on Self-play Methods in Reinforcement Learning","date":"2024-08-02","arxiv_id":"2408.01072","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01156","title":"TCR-GPT: Integrating Autoregressive Model and Reinforcement Learning for T-Cell Receptor Repertoires Generation","date":"2024-08-02","arxiv_id":"2408.01156","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01188","title":"Multi-Objective Deep Reinforcement Learning for Optimisation in Autonomous Systems","date":"2024-08-02","arxiv_id":"2408.01188","repositories_listed":0,"syntology":null},{"url":null,"slug":"2407-21359","title":"ProSpec RL: Plan Ahead, then Execute","date":"2024-07-31","arxiv_id":"2407.21359","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-transit-signal-priority-based-on","title":"Adaptive Transit Signal Priority based on Deep Reinforcement Learning and Connected Vehicles in a Traffic Microsimulation Environment","date":"2024-07-31","arxiv_id":"2408.00098","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-based-deep-reinforcement-learning-with","title":"Image-Based Deep Reinforcement Learning with Intrinsically Motivated Stimuli: On the Execution of Complex Robotic Tasks","date":"2024-07-31","arxiv_id":"2407.21338","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-assessment-with-qos-enhancement","title":"Multi-agent Assessment with QoS Enhancement for HD Map Updates in a Vehicular Network","date":"2024-07-31","arxiv_id":"2407.21460","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-generalizable-reinforcement-learning-1","title":"Towards Generalizable Reinforcement Learning via Causality-Guided Self-Adaptive Representations","date":"2024-07-30","arxiv_id":"2407.20651","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-method-for-fast-autonomy-transfer-in","title":"A Method for Fast Autonomy Transfer in Reinforcement Learning","date":"2024-07-29","arxiv_id":"2407.20466","repositories_listed":0,"syntology":null},{"url":null,"slug":"anomalous-state-sequence-modeling-to-enhance","title":"Anomalous State Sequence Modeling to Enhance Safety in Reinforcement Learning","date":"2024-07-29","arxiv_id":"2407.19860","repositories_listed":0,"syntology":null},{"url":null,"slug":"appraisal-guided-proximal-policy-optimization","title":"Appraisal-Guided Proximal Policy Optimization: Modeling Psychological Disorders in Dynamic Grid World","date":"2024-07-29","arxiv_id":"2407.20383","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolution-of-cooperation-in-the-public-goods","title":"Evolution of cooperation in the public goods game with Q-learning","date":"2024-07-29","arxiv_id":"2407.19851","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-provably-satisfy-high-relative","title":"Learning to Provably Satisfy High Relative Degree Constraints for Black-Box Systems","date":"2024-07-29","arxiv_id":"2407.20456","repositories_listed":0,"syntology":null},{"url":null,"slug":"qt-tdm-planning-with-transformer-dynamics","title":"QT-TDM: Planning With Transformer Dynamics Model and Autoregressive Q-Learning","date":"2024-07-26","arxiv_id":"2407.18841","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-anisotropic-p","title":"Reinforcement learning for anisotropic p-adaptation and error estimation in high-order solvers","date":"2024-07-26","arxiv_id":"2407.19000","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-quantum-architecture-search-in","title":"Differentiable Quantum Architecture Search in Asynchronous Quantum Reinforcement Learning","date":"2024-07-25","arxiv_id":"2407.18202","repositories_listed":0,"syntology":null},{"url":null,"slug":"path-following-and-stabilisation-of-a-bicycle","title":"Path Following and Stabilisation of a Bicycle Model using a Reinforcement Learning Approach","date":"2024-07-24","arxiv_id":"2407.17156","repositories_listed":0,"syntology":null},{"url":null,"slug":"pretrained-visual-representations-in","title":"Pretrained Visual Representations in Reinforcement Learning","date":"2024-07-24","arxiv_id":"2407.17238","repositories_listed":0,"syntology":null},{"url":null,"slug":"sonic-safe-social-navigation-with-adaptive","title":"SoNIC: Safe Social Navigation with Adaptive Conformal Inference and Constrained Reinforcement Learning","date":"2024-07-24","arxiv_id":"2407.17460","repositories_listed":0,"syntology":null},{"url":null,"slug":"sublinear-regret-for-an-actor-critic","title":"Sublinear Regret for a Class of Continuous-Time Linear-Quadratic Reinforcement Learning Problems","date":"2024-07-24","arxiv_id":"2407.17226","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-environment-shaping-is-the-next","title":"Automatic Environment Shaping is the Next Frontier in RL","date":"2024-07-23","arxiv_id":"2407.16186","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-imitation-to-refinement-residual-rl-for","title":"From Imitation to Refinement -- Residual RL for Precise Assembly","date":"2024-07-23","arxiv_id":"2407.16677","repositories_listed":0,"syntology":null},{"url":null,"slug":"odgr-online-dynamic-goal-recognition","title":"ODGR: Online Dynamic Goal Recognition","date":"2024-07-23","arxiv_id":"2407.16220","repositories_listed":0,"syntology":null},{"url":null,"slug":"secrm-2d-rl-based-efficient-and-comfortable","title":"SECRM-2D: RL-Based Efficient and Comfortable Route-Following Autonomous Driving with Analytic Safety Guarantees","date":"2024-07-23","arxiv_id":"2407.16857","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-intelligence-based-decision","title":"Artificial Intelligence-based Decision Support Systems for Precision and Digital Health","date":"2024-07-22","arxiv_id":"2407.16062","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-overview-of-reward-engineering","title":"Comprehensive Overview of Reward Engineering and Shaping in Advancing Reinforcement Learning Applications","date":"2024-07-22","arxiv_id":"2408.10215","repositories_listed":0,"syntology":null},{"url":null,"slug":"concept-based-interpretable-reinforcement","title":"Concept-Based Interpretable Reinforcement Learning with Limited to No Human Labels","date":"2024-07-22","arxiv_id":"2407.15786","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-sampling-guided-meta-training-for","title":"Importance Sampling-Guided Meta-Training for Intelligent Agents in Highly Interactive Environments","date":"2024-07-22","arxiv_id":"2407.15839","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-imitation-learning-through-graph","title":"Offline Imitation Learning Through Graph Search and Retrieval","date":"2024-07-22","arxiv_id":"2407.15403","repositories_listed":0,"syntology":null},{"url":null,"slug":"should-we-use-model-free-or-model-based","title":"Should we use model-free or model-based control? A case study of battery management systems","date":"2024-07-22","arxiv_id":"2407.15313","repositories_listed":0,"syntology":null},{"url":null,"slug":"rocket-landing-control-with-random-annealing","title":"Rocket Landing Control with Random Annealing Jump Start Reinforcement Learning","date":"2024-07-21","arxiv_id":"2407.15083","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-cell-populations-sharing","title":"Optimality theory of stigmergic collective information processing by chemotactic cells","date":"2024-07-21","arxiv_id":"2407.15298","repositories_listed":0,"syntology":null},{"url":null,"slug":"phase-re-service-in-reinforcement-learning","title":"Phase Re-service in Reinforcement Learning Traffic Signal Control","date":"2024-07-20","arxiv_id":"2407.14775","repositories_listed":0,"syntology":null},{"url":null,"slug":"fuzztherest-an-intelligent-automated-black","title":"FuzzTheREST: An Intelligent Automated Black-box RESTful API Fuzzer","date":"2024-07-19","arxiv_id":"2407.14361","repositories_listed":0,"syntology":null},{"url":null,"slug":"track-mdp-reinforcement-learning-for-target","title":"Track-MDP: Reinforcement Learning for Target Tracking with Controlled Sensing","date":"2024-07-19","arxiv_id":"2407.13995","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometric-active-exploration-in-markov","title":"Geometric Active Exploration in Markov Decision Processes: the Benefit of Abstraction","date":"2024-07-18","arxiv_id":"2407.13364","repositories_listed":0,"syntology":null},{"url":null,"slug":"random-latent-exploration-for-deep","title":"Random Latent Exploration for Deep Reinforcement Learning","date":"2024-07-18","arxiv_id":"2407.13755","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-tutorial-and-survey","title":"Reinforcement Learning: Tutorial and Survey","date":"2024-07-18","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sparsity-based-safety-conservatism-for","title":"Sparsity-based Safety Conservatism for Constrained Offline Reinforcement Learning","date":"2024-07-17","arxiv_id":"2407.13006","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-graph-based-adversarial-imitation-learning","title":"A Graph-based Adversarial Imitation Learning Framework for Reliable & Realtime Fleet Scheduling in Urban Air Mobility","date":"2024-07-16","arxiv_id":"2407.12113","repositories_listed":0,"syntology":null},{"url":null,"slug":"deflated-dynamics-value-iteration","title":"Deflated Dynamics Value Iteration","date":"2024-07-15","arxiv_id":"2407.10454","repositories_listed":0,"syntology":null},{"url":null,"slug":"superpadl-scaling-language-directed-physics","title":"SuperPADL: Scaling Language-Directed Physics-Based Control with Progressive Supervised Distillation","date":"2024-07-15","arxiv_id":"2407.10481","repositories_listed":0,"syntology":null},{"url":null,"slug":"affordance-guided-reinforcement-learning-via","title":"Affordance-Guided Reinforcement Learning via Visual Prompting","date":"2024-07-14","arxiv_id":"2407.10341","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-symmetric-1","title":"Deep reinforcement learning with symmetric data augmentation applied for aircraft lateral attitude tracking control","date":"2024-07-13","arxiv_id":"2407.11077","repositories_listed":0,"syntology":null},{"url":null,"slug":"global-reinforcement-learning-beyond-linear","title":"Global Reinforcement Learning: Beyond Linear and Convex Rewards via Submodular Semi-gradient Methods","date":"2024-07-13","arxiv_id":"2407.09905","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-aware-reinforcement-learning","title":"Communication-Aware Reinforcement Learning for Cooperative Adaptive Cruise Control","date":"2024-07-12","arxiv_id":"2407.08964","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-nine-physics-engines-for","title":"A Review of Nine Physics Engines for Reinforcement Learning Research","date":"2024-07-11","arxiv_id":"2407.08590","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-performance-and-user-engagement-in","title":"Enhancing Performance and User Engagement in Everyday Stress Monitoring: A Context-Aware Active Reinforcement Learning Approach","date":"2024-07-11","arxiv_id":"2407.08215","repositories_listed":0,"syntology":null},{"url":null,"slug":"pid-accelerated-temporal-difference","title":"PID Accelerated Temporal Difference Algorithms","date":"2024-07-11","arxiv_id":"2407.08803","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-control-with-coarse-to-fine","title":"Continuous Control with Coarse-to-fine Reinforcement Learning","date":"2024-07-10","arxiv_id":"2407.07787","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-in-hand-translation-using-tactile","title":"Learning In-Hand Translation Using Tactile Skin With Shear and Normal Force Sensing","date":"2024-07-10","arxiv_id":"2407.07885","repositories_listed":0,"syntology":null}],"record_sha256":"1a60da82ff762c153c24125949ac4cf23b9d57634ce7af5777d5e19281808f67","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}