{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/54","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":54,"pages_in_order":152,"rows_per_page":100,"rows":[5301,5400],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/53","next":"/task/reinforcement-learning-1/papers/55","papers":[{"url":null,"slug":"quality-driven-curation-of-remote-sensing","title":"Quality-Driven Curation of Remote Sensing Vision-Language Data via Learned Scoring Models","date":"2025-03-02","arxiv_id":"2503.00743","repositories_listed":0,"syntology":null},{"url":null,"slug":"never-too-prim-to-swim-an-llm-enhanced-rl","title":"Never too Prim to Swim: An LLM-Enhanced RL-based Adaptive S-Surface Controller for AUVs under Extreme Sea Conditions","date":"2025-03-01","arxiv_id":"2503.00527","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-reinforcement-learning-for-virtual","title":"Scalable Reinforcement Learning for Virtual Machine Scheduling","date":"2025-03-01","arxiv_id":"2503.00537","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-the-benefit-of","title":"Towards Understanding the Benefit of Multitask Representation Learning in Decision Process","date":"2025-03-01","arxiv_id":"2503.00345","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-reinforcement-learning-for-state","title":"Adaptive Reinforcement Learning for State Avoidance in Discrete Event Systems","date":"2025-02-28","arxiv_id":"2503.00192","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-and-modular-network-on-non","title":"Hierarchical and Modular Network on Non-prehensile Manipulation in General Environments","date":"2025-02-28","arxiv_id":"2502.20843","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-dreaming-a-global-workspace","title":"Multimodal Dreaming: A Global Workspace Approach to World Model-Based Reinforcement Learning","date":"2025-02-28","arxiv_id":"2502.21142","repositories_listed":0,"syntology":null},{"url":null,"slug":"subtask-aware-visual-reward-learning-from","title":"Subtask-Aware Visual Reward Learning from Segmented Demonstrations","date":"2025-02-28","arxiv_id":"2502.20630","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-model-based-reinforcement","title":"Accelerating Model-Based Reinforcement Learning with State-Space World Models","date":"2025-02-27","arxiv_id":"2502.20168","repositories_listed":0,"syntology":null},{"url":null,"slug":"carplanner-consistent-auto-regressive","title":"CarPlanner: Consistent Auto-regressive Trajectory Planning for Large-scale Reinforcement Learning in Autonomous Driving","date":"2025-02-27","arxiv_id":"2502.19908","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-efficiency-of-a-deep","title":"Improving the Efficiency of a Deep Reinforcement Learning-Based Power Management System for HPC Clusters Using Curriculum Learning","date":"2025-02-27","arxiv_id":"2502.20348","repositories_listed":0,"syntology":null},{"url":null,"slug":"r1-t1-fully-incentivizing-translation","title":"R1-T1: Fully Incentivizing Translation Capability in LLMs via Reasoning Learning","date":"2025-02-27","arxiv_id":"2502.19735","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-gymnasium-a-unified-modular-benchmark","title":"Robust Gymnasium: A Unified Modular Benchmark for Robust Reinforcement Learning","date":"2025-02-27","arxiv_id":"2502.19652","repositories_listed":0,"syntology":null},{"url":null,"slug":"distill-not-only-data-but-also-rewards-can","title":"Distill Not Only Data but Also Rewards: Can Smaller Language Models Surpass Larger Ones?","date":"2025-02-26","arxiv_id":"2502.19557","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalist-world-model-pre-training-for","title":"Efficient Reinforcement Learning by Guiding Generalist World Models with Non-Curated Data","date":"2025-02-26","arxiv_id":"2502.19544","repositories_listed":0,"syntology":null},{"url":null,"slug":"error-related-potential-driven-reinforcement","title":"Error-related Potential driven Reinforcement Learning for adaptive Brain-Computer Interfaces","date":"2025-02-25","arxiv_id":"2502.18594","repositories_listed":0,"syntology":null},{"url":null,"slug":"fetchbot-object-fetching-in-cluttered-shelves","title":"FetchBot: Object Fetching in Cluttered Shelves via Zero-Shot Sim2Real","date":"2025-02-25","arxiv_id":"2502.17894","repositories_listed":0,"syntology":null},{"url":null,"slug":"swe-rl-advancing-llm-reasoning-via","title":"SWE-RL: Advancing LLM Reasoning via Reinforcement Learning on Open Software Evolution","date":"2025-02-25","arxiv_id":"2502.18449","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-perceptions-to-decisions-wildfire","title":"From Perceptions to Decisions: Wildfire Evacuation Decision Prediction with Behavioral Theory-informed LLMs","date":"2025-02-24","arxiv_id":"2502.17701","repositories_listed":0,"syntology":null},{"url":null,"slug":"humanoid-whole-body-locomotion-on-narrow","title":"Humanoid Whole-Body Locomotion on Narrow Terrain via Dynamic Balance and Reinforcement Learning","date":"2025-02-24","arxiv_id":"2502.17219","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-liquidity-aware-bond-yields-using","title":"Predicting Liquidity-Aware Bond Yields using Causal GANs and Deep Reinforcement Learning with LLM Evaluation","date":"2025-02-24","arxiv_id":"2502.17011","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-on-strategic-mining-in-blockchain-a","title":"Survey on Strategic Mining in Blockchain: A Reinforcement Learning Approach","date":"2025-02-24","arxiv_id":"2502.17307","repositories_listed":0,"syntology":null},{"url":null,"slug":"yes-q-learning-helps-offline-in-context-rl","title":"Yes, Q-learning Helps Offline In-Context RL","date":"2025-02-24","arxiv_id":"2502.17666","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-rl-through-classifier-models","title":"Ensemble RL through Classifier Models: Enhancing Risk-Return Trade-offs in Trading Strategies","date":"2025-02-23","arxiv_id":"2502.17518","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-dependency-dynamics-in-multi-agent","title":"Toward Dependency Dynamics in Multi-Agent Reinforcement Learning for Traffic Signal Control","date":"2025-02-23","arxiv_id":"2502.16608","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-autonomous-network-orchestration-framework","title":"An Autonomous Network Orchestration Framework Integrating Large Language Models with Continual Reinforcement Learning","date":"2025-02-22","arxiv_id":"2502.16198","repositories_listed":0,"syntology":null},{"url":null,"slug":"together-we-rise-optimizing-real-time-multi","title":"Together We Rise: Optimizing Real-Time Multi-Robot Task Allocation using Coordinated Heterogeneous Plays","date":"2025-02-22","arxiv_id":"2502.16079","repositories_listed":0,"syntology":null},{"url":"/paper/hyperspherical-normalization-for-scalable","slug":"hyperspherical-normalization-for-scalable","title":"Hyperspherical Normalization for Scalable Deep Reinforcement Learning","date":"2025-02-21","arxiv_id":"2502.15280","repositories_listed":0,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hyperspherical-normalization-for-scalable#ran","syntology_url":"https://syntology.ai/paper/2502.15280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15280"}},"official":null}},{"url":null,"slug":"the-evolving-landscape-of-llm-and-vlm","title":"The Evolving Landscape of LLM- and VLM-Integrated Reinforcement Learning","date":"2025-02-21","arxiv_id":"2502.15214","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-highly-efficient-low-weight","title":"Discovering highly efficient low-weight quantum error-correcting codes with reinforcement learning","date":"2025-02-20","arxiv_id":"2502.14372","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-reward-free-offline-data-a-case","title":"Learning from Reward-Free Offline Data: A Case for Planning with Latent Dynamics Models","date":"2025-02-20","arxiv_id":"2502.14819","repositories_listed":0,"syntology":null},{"url":null,"slug":"mlgym-a-new-framework-and-benchmark-for","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","date":"2025-02-20","arxiv_id":"2502.14499","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-ultrasound-image","title":"Reinforcement Learning for Ultrasound Image Analysis A Comprehensive Review of Advances and Applications","date":"2025-02-20","arxiv_id":"2502.14995","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-graph-attention","title":"Reinforcement Learning with Graph Attention for Routing and Wavelength Assignment with Lightpath Reuse","date":"2025-02-20","arxiv_id":"2502.14741","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-review-on-the-control-of-heat","title":"Comprehensive Review on the Control of Heat Pumps for Energy Flexibility in Distribution Networks","date":"2025-02-19","arxiv_id":"2502.14111","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-rl-mpc-for-demand-response","title":"Hierarchical RL-MPC for Demand Response Scheduling","date":"2025-02-19","arxiv_id":"2502.13714","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-gene-based-testing-for-antibiotic","title":"Optimizing Gene-Based Testing for Antibiotic Resistance Prediction","date":"2025-02-19","arxiv_id":"2502.14919","repositories_listed":0,"syntology":null},{"url":null,"slug":"sppd-self-training-with-process-preference","title":"SPPD: Self-training with Process Preference Learning Using Dynamic Value Margin","date":"2025-02-19","arxiv_id":"2502.13516","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-quantification-for-markov-chains","title":"Uncertainty quantification for Markov chains with application to temporal difference learning","date":"2025-02-19","arxiv_id":"2502.13822","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-sim-to-real-methods-in-rl","title":"A Survey of Sim-to-Real Methods in RL: Progress, Prospects and Challenges with Foundation Models","date":"2025-02-18","arxiv_id":"2502.13187","repositories_listed":0,"syntology":null},{"url":null,"slug":"demystifying-multilingual-chain-of-thought-in","title":"Demystifying Multilingual Chain-of-Thought in Process Reward Modeling","date":"2025-02-18","arxiv_id":"2502.12663","repositories_listed":0,"syntology":null},{"url":null,"slug":"epo-explicit-policy-optimization-for","title":"EPO: Explicit Policy Optimization for Strategic Reasoning in LLMs via Reinforcement Learning","date":"2025-02-18","arxiv_id":"2502.12486","repositories_listed":0,"syntology":null},{"url":null,"slug":"localescaper-a-weakly-supervised-framework","title":"LocalEscaper: A Weakly-supervised Framework with Regional Reconstruction for Scalable Neural TSP Solvers","date":"2025-02-18","arxiv_id":"2502.12484","repositories_listed":0,"syntology":null},{"url":null,"slug":"rad-training-an-end-to-end-driving-policy-via","title":"RAD: Training an End-to-End Driving Policy via Large-Scale 3DGS-based Reinforcement Learning","date":"2025-02-18","arxiv_id":"2502.13144","repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-moral-uncertainty-using-large","title":"Addressing Moral Uncertainty using Large Language Models for Ethical Decision-Making","date":"2025-02-17","arxiv_id":"2503.05724","repositories_listed":0,"syntology":null},{"url":null,"slug":"camel-continuous-action-masking-enabled-by","title":"CAMEL: Continuous Action Masking Enabled by Large Language Models for Reinforcement Learning","date":"2025-02-17","arxiv_id":"2502.11896","repositories_listed":0,"syntology":null},{"url":null,"slug":"fitlight-federated-imitation-learning-for","title":"FitLight: Federated Imitation Learning for Plug-and-Play Autonomous Traffic Signal Control","date":"2025-02-17","arxiv_id":"2502.11937","repositories_listed":0,"syntology":null},{"url":null,"slug":"hovering-flight-of-soft-actuated-insect-scale","title":"Hovering Flight of Soft-Actuated Insect-Scale Micro Aerial Vehicles using Deep Reinforcement Learning","date":"2025-02-17","arxiv_id":"2502.12355","repositories_listed":0,"syntology":null},{"url":null,"slug":"intersectional-fairness-in-reinforcement","title":"Intersectional Fairness in Reinforcement Learning with Large State and Constraint Spaces","date":"2025-02-17","arxiv_id":"2502.11828","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-plasma-dynamics-and-robust-rampdown","title":"Learning Plasma Dynamics and Robust Rampdown Trajectories with Predict-First Experiments at TCV","date":"2025-02-17","arxiv_id":"2502.12327","repositories_listed":0,"syntology":null},{"url":null,"slug":"robot-deformable-object-manipulation-via-nmpc","title":"Robot Deformable Object Manipulation via NMPC-generated Demonstrations in Deep Reinforcement Learning","date":"2025-02-17","arxiv_id":"2502.11375","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-test-time-compute-without","title":"Scaling Test-Time Compute Without Verification or RL is Suboptimal","date":"2025-02-17","arxiv_id":"2502.12118","repositories_listed":0,"syntology":null},{"url":null,"slug":"textsc-flag-trader-fusion-llm-agent-with","title":"FLAG-Trader: Fusion LLM-Agent with Gradient-based Reinforcement Learning for Financial Trading","date":"2025-02-17","arxiv_id":"2502.11433","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlp-vision-language-preference-learning-for","title":"VLP: Vision-Language Preference Learning for Embodied Manipulation","date":"2025-02-17","arxiv_id":"2502.11918","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-multi-agent-offline-reinforcement","title":"Scalable Multi-Agent Offline Reinforcement Learning and the Role of Information","date":"2025-02-16","arxiv_id":"2502.11260","repositories_listed":0,"syntology":null},{"url":null,"slug":"rule-bottleneck-reinforcement-learning-joint","title":"Rule-Bottleneck Reinforcement Learning: Joint Explanation and Decision Optimization for Resource Allocation with Language Agents","date":"2025-02-15","arxiv_id":"2502.10732","repositories_listed":0,"syntology":null},{"url":null,"slug":"tackling-the-zero-shot-reinforcement-learning","title":"Tackling the Zero-Shot Reinforcement Learning Loss Directly","date":"2025-02-15","arxiv_id":"2502.10792","repositories_listed":0,"syntology":null},{"url":null,"slug":"beamdojo-learning-agile-humanoid-locomotion","title":"BeamDojo: Learning Agile Humanoid Locomotion on Sparse Footholds","date":"2025-02-14","arxiv_id":"2502.10363","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-information-prioritization-for","title":"Causal Information Prioritization for Efficient Reinforcement Learning","date":"2025-02-14","arxiv_id":"2502.10097","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-reinforcement-learning-for-actors","title":"Dynamic Reinforcement Learning for Actors","date":"2025-02-14","arxiv_id":"2502.10200","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-rl-under-episode-wise","title":"Provably Efficient RL under Episode-Wise Safety in Constrained MDPs with Linear Function Approximation","date":"2025-02-14","arxiv_id":"2502.10138","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-strategy-based-and","title":"Reinforcement Learning in Strategy-Based and Atari Games: A Review of Google DeepMinds Innovations","date":"2025-02-14","arxiv_id":"2502.10303","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-reinforcement-learning-for","title":"A Survey of Reinforcement Learning for Optimization in Automation","date":"2025-02-13","arxiv_id":"2502.09417","repositories_listed":0,"syntology":null},{"url":null,"slug":"diverse-transformer-decoding-for-offline","title":"Diverse Transformer Decoding for Offline Reinforcement Learning Using Financial Algorithmic Approaches","date":"2025-02-13","arxiv_id":"2502.10473","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-based-control-for","title":"Safe Reinforcement Learning-based Control for Hydrogen Diesel Dual-Fuel Engines","date":"2025-02-13","arxiv_id":"2502.09826","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-real-to-sim-to-real-approach-to-robotic","title":"A Real-to-Sim-to-Real Approach to Robotic Manipulation with VLM-Generated Iterative Keypoint Rewards","date":"2025-02-12","arxiv_id":"2502.08643","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-data-centric-ai-tabular-learning","title":"A Survey on Data-Centric AI: Tabular Learning from Reinforcement Learning and Generative AI Perspective","date":"2025-02-12","arxiv_id":"2502.08828","repositories_listed":0,"syntology":null},{"url":null,"slug":"combo-grasp-learning-constraint-based","title":"COMBO-Grasp: Learning Constraint-Based Manipulation for Bimanual Occluded Grasping","date":"2025-02-12","arxiv_id":"2502.08054","repositories_listed":0,"syntology":null},{"url":"/paper/hierarchical-multi-agent-framework-for-carbon","slug":"hierarchical-multi-agent-framework-for-carbon","title":"Hierarchical Multi-Agent Framework for Carbon-Efficient Liquid-Cooled Data Center Clusters","date":"2025-02-12","arxiv_id":"2502.08337","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hierarchical-multi-agent-framework-for-carbon#ran","syntology_url":"https://syntology.ai/paper/2502.08337","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.08337"}},"official":null}},{"url":null,"slug":"necessary-and-sufficient-oracles-toward-a","title":"Necessary and Sufficient Oracles: Toward a Computational Taxonomy For Reinforcement Learning","date":"2025-02-12","arxiv_id":"2502.08632","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-in-context-reinforcement-learning","title":"A Survey of In-Context Reinforcement Learning","date":"2025-02-11","arxiv_id":"2502.07978","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploratory-diffusion-policy-for-unsupervised","title":"Exploratory Diffusion Model for Unsupervised Reinforcement Learning","date":"2025-02-11","arxiv_id":"2502.07279","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-selection-for-off-policy-evaluation-new","title":"Model Selection for Off-policy Evaluation: New Algorithms and Experimental Protocol","date":"2025-02-11","arxiv_id":"2502.08021","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-sample-complexity-in-reward-free","title":"Near-Optimal Sample Complexity in Reward-Free Kernel-Based Reinforcement Learning","date":"2025-02-11","arxiv_id":"2502.07715","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-actuator-attacks-on-autonomous","title":"Optimal Actuator Attacks on Autonomous Vehicles Using Reinforcement Learning","date":"2025-02-11","arxiv_id":"2502.07839","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-formal-theory-of-the-need-for","title":"Towards a Formal Theory of the Need for Competence via Computational Intrinsic Motivation","date":"2025-02-11","arxiv_id":"2502.07423","repositories_listed":0,"syntology":null},{"url":null,"slug":"vsc-rl-advancing-autonomous-vision-language","title":"Advancing Autonomous VLM Agents via Variational Subgoal-Conditioned Reinforcement Learning","date":"2025-02-11","arxiv_id":"2502.07949","repositories_listed":0,"syntology":null},{"url":null,"slug":"select-before-act-spatially-decoupled-action","title":"Select before Act: Spatially Decoupled Action Repetition for Continuous Control","date":"2025-02-10","arxiv_id":"2502.06919","repositories_listed":0,"syntology":null},{"url":null,"slug":"smell-of-source-learning-based-odor-source","title":"Smell of Source: Learning-Based Odor Source Localization with Molecular Communication","date":"2025-02-10","arxiv_id":"2502.07112","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-offloading-in-vehicular-edge-computing","title":"Intelligent Offloading in Vehicular Edge Computing: A Comprehensive Review of Deep Reinforcement Learning Approaches and Architectures","date":"2025-02-10","arxiv_id":"2502.06963","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-stochastic-combinatorial","title":"Sequential Stochastic Combinatorial Optimization Using Hierarchal Reinforcement Learning","date":"2025-02-08","arxiv_id":"2502.05537","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarially-robust-td-learning-with","title":"Adversarially-Robust TD Learning with Markovian Data: Finite-Time Rates and Fundamental Limits","date":"2025-02-07","arxiv_id":"2502.04662","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-regularized-diffusion-policy","title":"Behavior-Regularized Diffusion Policy Optimization for Offline Reinforcement Learning","date":"2025-02-07","arxiv_id":"2502.04778","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergent-nmpc-based-reinforcement-learning","title":"Convergent NMPC-based Reinforcement Learning Using Deep Expected Sarsa and Nonlinear Temporal Difference Learning","date":"2025-02-07","arxiv_id":"2502.04925","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-pre-trained-decision-transformers","title":"Enhancing Pre-Trained Decision Transformers with Prompt-Tuning Bandits","date":"2025-02-07","arxiv_id":"2502.04979","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-strategic-language-agents-in-the","title":"Learning Strategic Language Agents in the Werewolf Game with Iterative Latent Space Policy Optimization","date":"2025-02-07","arxiv_id":"2502.04686","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-smarter-sensing-2d-clutter-mitigation","title":"Towards Smarter Sensing: 2D Clutter Mitigation in RL-Driven Cognitive MIMO Radar","date":"2025-02-07","arxiv_id":"2502.04967","repositories_listed":0,"syntology":null},{"url":null,"slug":"autotelic-reinforcement-learning-exploring","title":"Autotelic Reinforcement Learning: Exploring Intrinsic Motivations for Skill Acquisition in Open-Ended Environments","date":"2025-02-06","arxiv_id":"2502.04418","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavioral-entropy-guided-dataset-generation","title":"Behavioral Entropy-Guided Dataset Generation for Offline Reinforcement Learning","date":"2025-02-06","arxiv_id":"2502.04141","repositories_listed":0,"syntology":null},{"url":null,"slug":"illuminating-spaces-deep-reinforcement","title":"Illuminating Spaces: Deep Reinforcement Learning and Laser-Wall Partitioning for Architectural Layout Generation","date":"2025-02-06","arxiv_id":"2502.04407","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-alignment-as-retriever-optimization-an","title":"LLM Alignment as Retriever Optimization: An Information Retrieval Perspective","date":"2025-02-06","arxiv_id":"2502.03699","repositories_listed":0,"syntology":null},{"url":null,"slug":"mirror-descent-actor-critic-via-bounded","title":"Mirror Descent Actor Critic via Bounded Advantage Learning","date":"2025-02-06","arxiv_id":"2502.03854","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-prediction-of","title":"Reinforcement Learning Based Prediction of PID Controller Gains for Quadrotor UAVs","date":"2025-02-06","arxiv_id":"2502.04552","repositories_listed":0,"syntology":null},{"url":null,"slug":"transforming-multimodal-models-into-action","title":"Transforming Multimodal Models into Action Models for Radiotherapy","date":"2025-02-06","arxiv_id":"2502.04408","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-driven-materials-design-a-mini-review","title":"AI-driven materials design: a mini-review","date":"2025-02-05","arxiv_id":"2502.02905","repositories_listed":0,"syntology":null},{"url":null,"slug":"calibrated-unsupervised-anomaly-detection-in","title":"Calibrated Unsupervised Anomaly Detection in Multivariate Time-series using Reinforcement Learning","date":"2025-02-05","arxiv_id":"2502.03245","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnirl-in-context-reinforcement-learning-by","title":"OmniRL: In-Context Reinforcement Learning by Large-Scale Meta-Training in Randomized Worlds","date":"2025-02-05","arxiv_id":"2502.02869","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-electric-vehicles-charging-using","title":"Optimizing Electric Vehicles Charging using Large Language Models and Graph Neural Networks","date":"2025-02-05","arxiv_id":"2502.03067","repositories_listed":0,"syntology":null},{"url":null,"slug":"adviser-actor-critic-eliminating-steady-state","title":"Adviser-Actor-Critic: Eliminating Steady-State Error in Reinforcement Learning Control","date":"2025-02-04","arxiv_id":"2502.02265","repositories_listed":0,"syntology":null},{"url":null,"slug":"brief-analysis-of-deepseek-r1-and-it-s","title":"Brief analysis of DeepSeek R1 and it's implications for Generative AI","date":"2025-02-04","arxiv_id":"2502.02523","repositories_listed":0,"syntology":null}],"record_sha256":"670412c7e5e2b4d2c1d3c83fd9f5d47022c7004fb52ee316d0d9702edb342836","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}