{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/53","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":53,"pages_in_order":152,"rows_per_page":100,"rows":[5201,5300],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/52","next":"/task/reinforcement-learning-1/papers/54","papers":[{"url":null,"slug":"the-crucial-role-of-problem-formulation-in","title":"The Crucial Role of Problem Formulation in Real-World Reinforcement Learning","date":"2025-03-26","arxiv_id":"2503.20442","repositories_listed":0,"syntology":null},{"url":null,"slug":"think-twice-enhancing-llm-reasoning-by","title":"Think Twice: Enhancing LLM Reasoning by Scaling Multi-round Test-time Thinking","date":"2025-03-25","arxiv_id":"2503.19855","repositories_listed":0,"syntology":null},{"url":null,"slug":"aed-automatic-discovery-of-effective-and","title":"AED: Automatic Discovery of Effective and Diverse Vulnerabilities for Autonomous Driving Policy with Large Language Models","date":"2025-03-24","arxiv_id":"2503.20804","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-policy-optimization","title":"Evolutionary Policy Optimization","date":"2025-03-24","arxiv_id":"2503.19037","repositories_listed":0,"syntology":null},{"url":null,"slug":"option-discovery-using-llm-guided-semantic","title":"Option Discovery Using LLM-guided Semantic Hierarchical Reinforcement Learning","date":"2025-03-24","arxiv_id":"2503.19007","repositories_listed":0,"syntology":null},{"url":null,"slug":"parental-guidance-efficient-lifelong-learning","title":"Parental Guidance: Efficient Lifelong Learning through Evolutionary Distillation","date":"2025-03-24","arxiv_id":"2503.18531","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlcad-reinforcement-learning-training-gym-for","title":"RLCAD: Reinforcement Learning Training Gym for Revolution Involved CAD Command Sequence Generation","date":"2025-03-24","arxiv_id":"2503.18549","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-of-2","title":"Sample-Efficient Reinforcement Learning of Koopman eNMPC","date":"2025-03-24","arxiv_id":"2503.18787","repositories_listed":0,"syntology":null},{"url":null,"slug":"teaching-llms-for-step-level-automatic-math","title":"Teaching LLMs for Step-Level Automatic Math Correction via Reinforcement Learning","date":"2025-03-24","arxiv_id":"2503.18432","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-multi-fidelity-reinforcement","title":"Adaptive Multi-Fidelity Reinforcement Learning for Variance Reduction in Engineering Design Optimization","date":"2025-03-23","arxiv_id":"2503.18229","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-reward-over-optimization-in-rlhf","title":"Mitigating Reward Over-Optimization in RLHF via Behavior-Supported Regularization","date":"2025-03-23","arxiv_id":"2503.18130","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-navigation-and-chemical","title":"Optimizing Navigation And Chemical Application in Precision Agriculture With Deep Reinforcement Learning And Conditional Action Tree","date":"2025-03-23","arxiv_id":"2503.17985","repositories_listed":0,"syntology":null},{"url":null,"slug":"viva-video-trained-value-functions-for","title":"ViVa: Video-Trained Value Functions for Guiding Online RL from Diverse Data","date":"2025-03-23","arxiv_id":"2503.18210","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-roadmap-towards-improving-multi-agent","title":"A Roadmap Towards Improving Multi-Agent Reinforcement Learning With Causal Discovery And Inference","date":"2025-03-22","arxiv_id":"2503.17803","repositories_listed":0,"syntology":null},{"url":null,"slug":"comfygpt-a-self-optimizing-multi-agent-system","title":"ComfyGPT: A Self-Optimizing Multi-Agent System for Comprehensive ComfyUI Workflow Generation","date":"2025-03-22","arxiv_id":"2503.17671","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferable-latent-to-latent-locomotion","title":"Transferable Latent-to-Latent Locomotion Policy for Efficient and Versatile Motion Control of Diverse Legged Robots","date":"2025-03-22","arxiv_id":"2503.17626","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-radiotherapy-treatment-planning","title":"Autonomous Radiotherapy Treatment Planning Using DOLA: A Privacy-Preserving, LLM-Based Optimization Agent","date":"2025-03-21","arxiv_id":"2503.17553","repositories_listed":0,"syntology":null},{"url":"/paper/causally-aligned-curriculum-learning","slug":"causally-aligned-curriculum-learning","title":"Causally Aligned Curriculum Learning","date":"2025-03-21","arxiv_id":"2503.16799","repositories_listed":0,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/causally-aligned-curriculum-learning#ran","syntology_url":"https://syntology.ai/paper/2503.16799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16799"}},"official":null}},{"url":null,"slug":"grammar-and-gameplay-aligned-rl-for-game","title":"Grammar and Gameplay-aligned RL for Game Description Generation with LLMs","date":"2025-03-20","arxiv_id":"2503.15783","repositories_listed":0,"syntology":null},{"url":null,"slug":"othink-mr1-stimulating-multimodal-generalized","title":"OThink-MR1: Stimulating multimodal generalized reasoning capabilities via dynamic reinforcement learning","date":"2025-03-20","arxiv_id":"2503.16081","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl4med-ddpo-reinforcement-learning-for","title":"RL4Med-DDPO: Reinforcement Learning for Controlled Guidance Towards Diverse Medical Image Generation using Vision-Language Foundation Models","date":"2025-03-20","arxiv_id":"2503.15784","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-automated-semantic-interpretability","title":"Towards Automated Semantic Interpretability in Reinforcement Learning via Vision-Language Models","date":"2025-03-20","arxiv_id":"2503.16724","repositories_listed":0,"syntology":null},{"url":null,"slug":"uas-visual-navigation-in-large-and-unseen","title":"UAS Visual Navigation in Large and Unseen Environments via a Meta Agent","date":"2025-03-20","arxiv_id":"2503.15781","repositories_listed":0,"syntology":null},{"url":null,"slug":"1000-layer-networks-for-self-supervised-rl","title":"1000 Layer Networks for Self-Supervised RL: Scaling Depth Can Enable New Goal-Reaching Capabilities","date":"2025-03-19","arxiv_id":"2503.14858","repositories_listed":0,"syntology":null},{"url":null,"slug":"behaviour-discovery-and-attribution-for","title":"Behaviour Discovery and Attribution for Explainable Reinforcement Learning","date":"2025-03-19","arxiv_id":"2503.14973","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-review-of-reinforcement","title":"Comprehensive Review of Reinforcement Learning for Medical Ultrasound Imaging","date":"2025-03-19","arxiv_id":"2503.16543","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepmesh-auto-regressive-artist-mesh-creation","title":"DeepMesh: Auto-Regressive Artist-mesh Creation with Reinforcement Learning","date":"2025-03-19","arxiv_id":"2503.15265","repositories_listed":0,"syntology":null},{"url":null,"slug":"empowering-medical-multi-agents-with-clinical","title":"Empowering Medical Multi-Agents with Clinical Consultation Flow for Dynamic Diagnosis","date":"2025-03-19","arxiv_id":"2503.16547","repositories_listed":0,"syntology":null},{"url":null,"slug":"good-actions-succeed-bad-actions-generalize-a","title":"Good Actions Succeed, Bad Actions Generalize: A Case Study on Why RL Generalizes Better","date":"2025-03-19","arxiv_id":"2503.15693","repositories_listed":0,"syntology":null},{"url":null,"slug":"logllama-transformer-based-log-anomaly","title":"LogLLaMA: Transformer-based log anomaly detection with LLaMA","date":"2025-03-19","arxiv_id":"2503.14849","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-environment-with-llm","title":"Reinforcement Learning Environment with LLM-Controlled Adversary in D&D 5th Edition Combat","date":"2025-03-19","arxiv_id":"2503.15726","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-training-wheels-adaptive-auxiliary","title":"Reward Training Wheels: Adaptive Auxiliary Rewards for Robotics Reinforcement Learning","date":"2025-03-19","arxiv_id":"2503.15724","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctsac-curriculum-based-transformer-soft-actor","title":"CTSAC: Curriculum-Based Transformer Soft Actor-Critic for Goal-Oriented Robot Exploration","date":"2025-03-18","arxiv_id":"2503.14254","repositories_listed":0,"syntology":null},{"url":null,"slug":"pauli-network-circuit-synthesis-with","title":"Pauli Network Circuit Synthesis with Reinforcement Learning","date":"2025-03-18","arxiv_id":"2503.14448","repositories_listed":0,"syntology":null},{"url":null,"slug":"revealing-higher-order-neural-representations","title":"Revealing higher-order neural representations of uncertainty with the Noise Estimation through Reinforcement-based Diffusion (NERD) model","date":"2025-03-18","arxiv_id":"2503.14333","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-driven-transformer","title":"A Reinforcement Learning-Driven Transformer GAN for Molecular Generation","date":"2025-03-17","arxiv_id":"2503.12796","repositories_listed":0,"syntology":null},{"url":null,"slug":"apf-boosting-adaptive-potential-function","title":"APF+: Boosting adaptive-potential function reinforcement learning methods with a W-shaped network for high-dimensional games","date":"2025-03-17","arxiv_id":"2503.13557","repositories_listed":0,"syntology":null},{"url":null,"slug":"flex-a-framework-for-learning-robot-agnostic","title":"FLEX: A Framework for Learning Robot-Agnostic Force-based Skills Involving Sustained Contact Object Manipulation","date":"2025-03-17","arxiv_id":"2503.13418","repositories_listed":0,"syntology":null},{"url":null,"slug":"synchronous-vs-asynchronous-reinforcement","title":"Synchronous vs Asynchronous Reinforcement Learning in a Real World Robot","date":"2025-03-17","arxiv_id":"2503.14554","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-angle-selection-in-x-ray-ct-a","title":"Dynamic Angle Selection in X-Ray CT: A Reinforcement Learning Approach to Optimal Stopping","date":"2025-03-16","arxiv_id":"2503.12688","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-time-policy-switching-for-offline","title":"Evaluation-Time Policy Switching for Offline Reinforcement Learning","date":"2025-03-15","arxiv_id":"2503.12222","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-torque-control-of-exoskeletons-under","title":"Adaptive Torque Control of Exoskeletons under Spasticity Conditions via Reinforcement Learning","date":"2025-03-14","arxiv_id":"2503.11433","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-obstacle-avoidance-with-bounded","title":"Dynamic Obstacle Avoidance with Bounded Rationality Adversarial Reinforcement Learning","date":"2025-03-14","arxiv_id":"2503.11467","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-competitive-and-collusive-behaviors","title":"Exploring Competitive and Collusive Behaviors in Algorithmic Pricing with Deep Reinforcement Learning","date":"2025-03-14","arxiv_id":"2503.11270","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-controlled","title":"Reinforcement Learning-Based Controlled Switching Approach for Inrush Current Minimization in Power Transformers","date":"2025-03-14","arxiv_id":"2503.11398","repositories_listed":0,"syntology":null},{"url":null,"slug":"sketch-to-skill-bootstrapping-robot-learning","title":"Sketch-to-Skill: Bootstrapping Robot Learning with Human Drawn Trajectory Sketches","date":"2025-03-14","arxiv_id":"2503.11918","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-mutual-empowerment-between-wireless","title":"DeepSeek-Inspired Exploration of RL-based LLMs and Synergy with Wireless Networks: A Survey","date":"2025-03-13","arxiv_id":"2503.09956","repositories_listed":0,"syntology":null},{"url":null,"slug":"h2-marl-multi-agent-reinforcement-learning","title":"H2-MARL: Multi-Agent Reinforcement Learning for Pareto Optimality in Hospital Capacity Strain and Human Mobility during Epidemic","date":"2025-03-13","arxiv_id":"2503.10907","repositories_listed":0,"syntology":null},{"url":null,"slug":"nil-no-data-imitation-learning-by-leveraging","title":"NIL: No-data Imitation Learning by Leveraging Pre-trained Video Diffusion Models","date":"2025-03-13","arxiv_id":"2503.10626","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-based-reward-modeling-for","title":"Representation-based Reward Modeling for Efficient Safety Alignment of Large Language Model","date":"2025-03-13","arxiv_id":"2503.10093","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-continual-domain-adaptation-after","title":"Safe Continual Domain Adaptation after Sim2Real Transfer of Reinforcement Learning Policies in Robotics","date":"2025-03-13","arxiv_id":"2503.10949","repositories_listed":0,"syntology":null},{"url":null,"slug":"sortingenv-an-extendable-rl-environment-for","title":"SortingEnv: An Extendable RL-Environment for an Industrial Sorting Process","date":"2025-03-13","arxiv_id":"2503.10466","repositories_listed":0,"syntology":null},{"url":null,"slug":"sysllm-generating-synthesized-policy","title":"SySLLM: Generating Synthesized Policy Summaries for Reinforcement Learning Agents Using Large Language Models","date":"2025-03-13","arxiv_id":"2503.10509","repositories_listed":0,"syntology":null},{"url":null,"slug":"edge-ai-powered-real-time-decision-making-for","title":"Edge AI-Powered Real-Time Decision-Making for Autonomous Vehicles in Adverse Weather Conditions","date":"2025-03-12","arxiv_id":"2503.09638","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-reinforcement-learning-safety-and","title":"Evaluating Reinforcement Learning Safety and Trustworthiness in Cyber-Physical Systems","date":"2025-03-12","arxiv_id":"2503.09388","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-regional-traffic-signal-control","title":"Large-scale Regional Traffic Signal Control Based on Single-Agent Reinforcement Learning","date":"2025-03-12","arxiv_id":"2503.09252","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-look-ahead-guidance-via-verifier-in-the","title":"Local Look-Ahead Guidance via Verifier-in-the-Loop for Automated Theorem Proving","date":"2025-03-12","arxiv_id":"2503.09730","repositories_listed":0,"syntology":null},{"url":null,"slug":"marinegym-a-high-performance-reinforcement","title":"MarineGym: A High-Performance Reinforcement Learning Platform for Underwater Robotics","date":"2025-03-12","arxiv_id":"2503.09203","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimisation-of-the-accelerator-control-by","title":"Optimisation of the Accelerator Control by Reinforcement Learning: A Simulation-Based Approach","date":"2025-03-12","arxiv_id":"2503.09665","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-bayesian-inverse-problems-with-1","title":"Solving Bayesian inverse problems with diffusion priors and off-policy RL","date":"2025-03-12","arxiv_id":"2503.09746","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-locomotion-transformer-with","title":"Unified Locomotion Transformer with Simultaneous Sim-to-Real Transfer for Quadrupeds","date":"2025-03-12","arxiv_id":"2503.08997","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cascading-cooperative-multi-agent-framework","title":"A Cascading Cooperative Multi-agent Framework for On-ramp Merging Control Integrating Large Language Models","date":"2025-03-11","arxiv_id":"2503.08199","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-soc-in-battery-cells-using-safe","title":"Balancing SoC in Battery Cells using Safe Action Perturbations","date":"2025-03-11","arxiv_id":"2503.11696","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-world-models-learning-to","title":"Disentangled World Models: Learning to Transfer Semantic Knowledge from Distracting Videos for Reinforcement Learning","date":"2025-03-11","arxiv_id":"2503.08751","repositories_listed":0,"syntology":null},{"url":null,"slug":"hasard-a-benchmark-for-vision-based-safe","title":"HASARD: A Benchmark for Vision-Based Safe Reinforcement Learning in Embodied Agents","date":"2025-03-11","arxiv_id":"2503.08241","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-prospect-and-retrospect-reflective-memory","title":"In Prospect and Retrospect: Reflective Memory Management for Long-term Personalized Dialogue Agents","date":"2025-03-11","arxiv_id":"2503.08026","repositories_listed":0,"syntology":null},{"url":null,"slug":"more-unlocking-scalability-in-reinforcement","title":"MoRE: Unlocking Scalability in Reinforcement Learning for Quadruped Vision-Language-Action Models","date":"2025-03-11","arxiv_id":"2503.08007","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-sample-complexity-for-iterated","title":"Near-Optimal Sample Complexity for Iterated CVaR Reinforcement Learning with a Generative Model","date":"2025-03-11","arxiv_id":"2503.08934","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-action-generalization-with-limited","title":"Zero-Shot Action Generalization with Limited Observations","date":"2025-03-11","arxiv_id":"2503.08867","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-routing-protocols-for-determining","title":"Adaptive routing protocols for determining optimal paths in AI multi-agent systems: a priority- and learning-enhanced approach","date":"2025-03-10","arxiv_id":"2503.07686","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-neural-clause-selection","title":"Efficient Neural Clause-Selection Reinforcement","date":"2025-03-10","arxiv_id":"2503.07792","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-test-time-compute-via-meta","title":"Optimizing Test-Time Compute via Meta Reinforcement Fine-Tuning","date":"2025-03-10","arxiv_id":"2503.07572","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-multi-objective-reinforcement","title":"A Novel Multi-Objective Reinforcement Learning Algorithm for Pursuit-Evasion Game","date":"2025-03-09","arxiv_id":"2503.06741","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-load-balancing-for-ev-charging","title":"Dynamic Load Balancing for EV Charging Stations Using Reinforcement Learning and Demand Prediction","date":"2025-03-09","arxiv_id":"2503.06370","repositories_listed":0,"syntology":null},{"url":null,"slug":"gflowvlm-enhancing-multi-step-reasoning-in","title":"GFlowVLM: Enhancing Multi-step Reasoning in Vision-Language Models with Generative Flow Networks","date":"2025-03-09","arxiv_id":"2503.06514","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-shielding-for-safe","title":"Probabilistic Shielding for Safe Reinforcement Learning","date":"2025-03-09","arxiv_id":"2503.07671","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-assisted-coverage-hole-detection-using","title":"UAV-Assisted Coverage Hole Detection Using Reinforcement Learning in Urban Cellular Networks","date":"2025-03-09","arxiv_id":"2503.06494","repositories_listed":0,"syntology":null},{"url":null,"slug":"synergizing-ai-and-digital-twins-for-next","title":"Synergizing AI and Digital Twins for Next-Generation Network Optimization, Forecasting, and Security","date":"2025-03-08","arxiv_id":"2503.06302","repositories_listed":0,"syntology":null},{"url":null,"slug":"ultho-ultra-lightweight-yet-efficient","title":"ULTHO: Ultra-Lightweight yet Efficient Hyperparameter Optimization in Deep Reinforcement Learning","date":"2025-03-08","arxiv_id":"2503.06101","repositories_listed":0,"syntology":null},{"url":null,"slug":"vairiational-stochastic-games","title":"Vairiational Stochastic Games","date":"2025-03-08","arxiv_id":"2503.06037","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-multi-agent-q-learning-for-policy","title":"Generative Multi-Agent Q-Learning for Policy Optimization: Decentralized Wireless Networks","date":"2025-03-07","arxiv_id":"2503.05970","repositories_listed":0,"syntology":null},{"url":null,"slug":"guaranteeing-out-of-distribution-detection-in","title":"Guaranteeing Out-Of-Distribution Detection in Deep RL via Transition Estimation","date":"2025-03-07","arxiv_id":"2503.05238","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-fidelity-policy-gradient-algorithms","title":"Multi-Fidelity Policy Gradient Algorithms","date":"2025-03-07","arxiv_id":"2503.05696","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-robot-collaboration-through","title":"Multi-Robot Collaboration through Reinforcement Learning and Abstract Simulation","date":"2025-03-07","arxiv_id":"2503.05092","repositories_listed":0,"syntology":null},{"url":null,"slug":"tractable-representations-for-convergent","title":"Tractable Representations for Convergent Approximation of Distributional HJB Equations","date":"2025-03-07","arxiv_id":"2503.05563","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-optimize-deep-rl-policy-weights-as","title":"Can We Optimize Deep RL Policy Weights as Trajectory Modeling?","date":"2025-03-06","arxiv_id":"2503.04074","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-learning-from-human","title":"Data-Efficient Learning from Human Interventions for Mobile Robots","date":"2025-03-06","arxiv_id":"2503.04969","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-weighted-flow-matching-for-offline","title":"Energy-Weighted Flow Matching for Offline Reinforcement Learning","date":"2025-03-06","arxiv_id":"2503.04975","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-correct-automata-embeddings-for","title":"Provably Correct Automata Embeddings for Optimal Automata-Conditioned Reinforcement Learning","date":"2025-03-06","arxiv_id":"2503.05042","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-autonomous-reinforcement-learning-for","title":"Towards Autonomous Reinforcement Learning for Real-World Robotic Manipulation with Large Language Models","date":"2025-03-06","arxiv_id":"2503.04280","repositories_listed":0,"syntology":null},{"url":null,"slug":"rebalanced-multimodal-learning-with-data","title":"Rebalanced Multimodal Learning with Data-aware Unimodal Sampling","date":"2025-03-05","arxiv_id":"2503.03792","repositories_listed":0,"syntology":null},{"url":null,"slug":"dreamerv3-for-traffic-signal-control","title":"DreamerV3 for Traffic Signal Control: Hyperparameter Tuning and Performance","date":"2025-03-04","arxiv_id":"2503.02279","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantitative-resilience-modeling-for","title":"Quantitative Resilience Modeling for Autonomous Cyber Defense","date":"2025-03-04","arxiv_id":"2503.02780","repositories_listed":0,"syntology":null},{"url":null,"slug":"rewarding-doubt-a-reinforcement-learning","title":"Rewarding Doubt: A Reinforcement Learning Approach to Confidence Calibration of Large Language Models","date":"2025-03-04","arxiv_id":"2503.02623","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-01734","title":"Adversarial Agents: Black-Box Evasion Attacks with Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.01734","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-multi-task-temporal-difference","title":"Accelerating Multi-Task Temporal Difference Learning under Low-Rank Representation","date":"2025-03-03","arxiv_id":"2503.02030","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-alignments-of-lens-systems-with","title":"Active Alignments of Lens Systems with Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.02075","repositories_listed":0,"syntology":null},{"url":null,"slug":"all-roads-lead-to-likelihood-the-value-of","title":"All Roads Lead to Likelihood: The Value of Reinforcement Learning in Fine-Tuning","date":"2025-03-03","arxiv_id":"2503.01067","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-s-behind-ppo-s-collapse-in-long-cot","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","date":"2025-03-03","arxiv_id":"2503.01491","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-optimal-reinforcement-learning-with","title":"Minimax Optimal Reinforcement Learning with Quasi-Optimism","date":"2025-03-02","arxiv_id":"2503.00810","repositories_listed":0,"syntology":null}],"record_sha256":"afe11d2db75270ad52ab297dffe1e77254eb46cc738a3cb20486f798ec66164a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}