{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/52","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":52,"pages_in_order":152,"rows_per_page":100,"rows":[5101,5200],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/51","next":"/task/reinforcement-learning-1/papers/53","papers":[{"url":null,"slug":"hybrid-reinforcement-learning-and-model","title":"Hybrid Reinforcement Learning and Model Predictive Control for Adaptive Control of Hydrogen-Diesel Dual-Fuel Combustion","date":"2025-04-23","arxiv_id":"2504.16875","repositories_listed":0,"syntology":null},{"url":null,"slug":"monte-carlo-planning-with-large-language","title":"Monte Carlo Planning with Large Language Model for Text-Based Game Agents","date":"2025-04-23","arxiv_id":"2504.16855","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-policy-gradient-for-average-reward","title":"Natural Policy Gradient for Average Reward Non-Stationary RL","date":"2025-04-23","arxiv_id":"2504.16415","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-robotic-world-model-learning-robotic","title":"Offline Robotic World Model: Learning Robotic Policies without a Physics Simulator","date":"2025-04-23","arxiv_id":"2504.16680","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-framework-for-the","title":"Reinforcement learning framework for the mechanical design of microelectronic components under multiphysics constraints","date":"2025-04-23","arxiv_id":"2504.17142","repositories_listed":0,"syntology":null},{"url":null,"slug":"insights-from-verification-training-a-verilog","title":"Insights from Verification: Training a Verilog Generation LLM with Reinforcement Learning with Testbench Feedback","date":"2025-04-22","arxiv_id":"2504.15804","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-based-radiative-transfer-solving-the-2","title":"Policy-Based Radiative Transfer: Solving the $2$-Level Atom Non-LTE Problem using Soft Actor-Critic Reinforcement Learning","date":"2025-04-22","arxiv_id":"2504.15679","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-optimal-design-of-experiment-for","title":"Real-Time Optimal Design of Experiment for Parameter Identification of Li-Ion Cell Electrochemical Model","date":"2025-04-22","arxiv_id":"2504.15578","repositories_listed":0,"syntology":null},{"url":null,"slug":"sari-structured-audio-reasoning-via","title":"SARI: Structured Audio Reasoning via Curriculum-Guided Reinforcement Learning","date":"2025-04-22","arxiv_id":"2504.15900","repositories_listed":0,"syntology":null},{"url":null,"slug":"slim-gym-reinforcement-learning-for","title":"SLiM-Gym: Reinforcement Learning for Population Genetics","date":"2025-04-22","arxiv_id":"2504.16301","repositories_listed":0,"syntology":null},{"url":null,"slug":"streamrl-scalable-heterogeneous-and-elastic","title":"StreamRL: Scalable, Heterogeneous, and Elastic RL for LLMs with Disaggregated Stream Generation","date":"2025-04-22","arxiv_id":"2504.15930","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-contrastive-skill-learning-with-state","title":"Dynamic Contrastive Skill Learning with State-Transition Based Skill Clustering and Dynamic Length Adjustment","date":"2025-04-21","arxiv_id":"2504.14805","repositories_listed":0,"syntology":null},{"url":null,"slug":"lapp-large-language-model-feedback-for","title":"LAPP: Large Language Model Feedback for Preference-Driven Reinforcement Learning","date":"2025-04-21","arxiv_id":"2504.15472","repositories_listed":0,"syntology":null},{"url":null,"slug":"otc-optimal-tool-calls-via-reinforcement","title":"OTC: Optimal Tool Calls via Reinforcement Learning","date":"2025-04-21","arxiv_id":"2504.14870","repositories_listed":0,"syntology":null},{"url":null,"slug":"think2sql-reinforce-llm-reasoning","title":"Think2SQL: Reinforce LLM Reasoning Capabilities for Text2SQL","date":"2025-04-21","arxiv_id":"2504.15077","repositories_listed":0,"syntology":null},{"url":null,"slug":"relation-r1-cognitive-chain-of-thought-guided","title":"Relation-R1: Cognitive Chain-of-Thought Guided Reinforcement Learning for Unified Relational Comprehension","date":"2025-04-20","arxiv_id":"2504.14642","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rl-exploration-for-llm-reasoning","title":"Improving RL Exploration for LLM Reasoning through Retrospective Replay","date":"2025-04-19","arxiv_id":"2504.14363","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixed-precision-conjugate-gradient-solvers","title":"Mixed-Precision Conjugate Gradient Solvers with RL-Driven Precision Tuning","date":"2025-04-19","arxiv_id":"2504.14268","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-enhanced-reinforcement-learning-for-1","title":"Quantum-Enhanced Reinforcement Learning for Power Grid Security Assessment","date":"2025-04-19","arxiv_id":"2504.14412","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlearning-works-better-than-you-think-local","title":"Unlearning Works Better Than You Think: Local Reinforcement-Based Selection of Auxiliary Objectives","date":"2025-04-19","arxiv_id":"2504.14418","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-generalization-in-intent-detection","title":"Improving Generalization in Intent Detection: GRPO with Reward-Based Curriculum Sampling","date":"2025-04-18","arxiv_id":"2504.13592","repositories_listed":0,"syntology":null},{"url":null,"slug":"not-all-rollouts-are-useful-down-sampling","title":"Not All Rollouts are Useful: Down-Sampling Rollouts in LLM Reinforcement Learning","date":"2025-04-18","arxiv_id":"2504.13818","repositories_listed":0,"syntology":null},{"url":null,"slug":"switchmt-an-adaptive-context-switching","title":"SwitchMT: An Adaptive Context Switching Methodology for Scalable Multi-Task Learning in Intelligent Autonomous Agents","date":"2025-04-18","arxiv_id":"2504.13541","repositories_listed":0,"syntology":null},{"url":null,"slug":"crossing-the-human-robot-embodiment-gap-with","title":"Crossing the Human-Robot Embodiment Gap with Sim-to-Real RL using One Human Demonstration","date":"2025-04-17","arxiv_id":"2504.12609","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-policy-optimization-1","title":"Evolutionary Policy Optimization","date":"2025-04-17","arxiv_id":"2504.12568","repositories_listed":0,"syntology":null},{"url":null,"slug":"llms-meet-finance-fine-tuning-foundation","title":"LLMs Meet Finance: Fine-Tuning Foundation Models for the Open FinLLM Leaderboard","date":"2025-04-17","arxiv_id":"2504.13125","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-pinns-reinforcement-learning-driven","title":"RL-PINNs: Reinforcement Learning-Driven Adaptive Sampling for Efficient Training of PINNs","date":"2025-04-17","arxiv_id":"2504.12949","repositories_listed":0,"syntology":null},{"url":null,"slug":"traces-trajectory-based-credit-assignment","title":"TraCeS: Trajectory Based Credit Assignment From Sparse Safety Feedback","date":"2025-04-17","arxiv_id":"2504.12557","repositories_listed":0,"syntology":null},{"url":null,"slug":"d1-scaling-reasoning-in-diffusion-large","title":"d1: Scaling Reasoning in Diffusion Large Language Models via Reinforcement Learning","date":"2025-04-16","arxiv_id":"2504.12216","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-reinforcement-learning-for-1","title":"Evolutionary Reinforcement Learning for Interpretable Decision-Making in Supply Chain Management","date":"2025-04-16","arxiv_id":"2504.12023","repositories_listed":0,"syntology":null},{"url":null,"slug":"pix2pockets-shot-suggestions-in-8-ball-pool","title":"pix2pockets: Shot Suggestions in 8-Ball Pool from a Single Image in the Wild","date":"2025-04-16","arxiv_id":"2504.12045","repositories_listed":0,"syntology":null},{"url":"/paper/vipo-value-function-inconsistency-penalized","slug":"vipo-value-function-inconsistency-penalized","title":"VIPO: Value Function Inconsistency Penalized Offline Reinforcement Learning","date":"2025-04-16","arxiv_id":"2504.11944","repositories_listed":0,"syntology":{"n":14,"n_ran":10,"n_constructed":9,"n_ran_checked":9,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":14,"phrase":"10 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vipo-value-function-inconsistency-penalized#ran","syntology_url":"https://syntology.ai/paper/2504.11944","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11944"}},"official":null}},{"url":null,"slug":"achieving-tighter-finite-time-rates-for","title":"Achieving Tighter Finite-Time Rates for Heterogeneous Federated Stochastic Approximation under Markovian Sampling","date":"2025-04-15","arxiv_id":"2504.11645","repositories_listed":0,"syntology":null},{"url":null,"slug":"hallucination-aware-generative-pretrained","title":"Hallucination-Aware Generative Pretrained Transformer for Cooperative Aerial Mobility Control","date":"2025-04-15","arxiv_id":"2504.10831","repositories_listed":0,"syntology":null},{"url":null,"slug":"next-future-sample-efficient-policy-learning","title":"Next-Future: Sample-Efficient Policy Learning for Robotic-Arm Tasks","date":"2025-04-15","arxiv_id":"2504.11247","repositories_listed":0,"syntology":null},{"url":null,"slug":"position-paper-rethinking-privacy-in-rl-for","title":"Position Paper: Rethinking Privacy in RL for Sequential Decision-making in the Age of LLMs","date":"2025-04-15","arxiv_id":"2504.11511","repositories_listed":0,"syntology":null},{"url":null,"slug":"revealing-covert-attention-by-analyzing-human","title":"Revealing Covert Attention by Analyzing Human and Reinforcement Learning Agent Gameplay","date":"2025-04-15","arxiv_id":"2504.11118","repositories_listed":0,"syntology":null},{"url":null,"slug":"rezero-enhancing-llm-search-ability-by-trying","title":"ReZero: Enhancing LLM search ability by trying one-more-time","date":"2025-04-15","arxiv_id":"2504.11001","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-insurance-reserving-with-cvar","title":"Adaptive Insurance Reserving with CVaR-Constrained Reinforcement Learning under Macroeconomic Regimes","date":"2025-04-13","arxiv_id":"2504.09396","repositories_listed":0,"syntology":null},{"url":null,"slug":"cheatagent-attacking-llm-empowered","title":"CheatAgent: Attacking LLM-Empowered Recommender Systems via LLM Agent","date":"2025-04-13","arxiv_id":"2504.13192","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-implementation-of-reinforcement","title":"Efficient Implementation of Reinforcement Learning over Homomorphic Encryption","date":"2025-04-12","arxiv_id":"2504.09335","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-more-efficient-robust-instance","title":"Towards More Efficient, Robust, Instance-adaptive, and Generalizable Sequential Decision making","date":"2025-04-12","arxiv_id":"2504.09192","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-optimal-differentially-private-regret","title":"Towards Optimal Differentially Private Regret Bounds in Linear MDPs","date":"2025-04-12","arxiv_id":"2504.09339","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-distributional-learning-with-non","title":"Deep Distributional Learning with Non-crossing Quantile Network","date":"2025-04-11","arxiv_id":"2504.08215","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-normalization-for-lipschitz","title":"Spectral Normalization for Lipschitz-Constrained Policies on Learning Humanoid Locomotion","date":"2025-04-11","arxiv_id":"2504.08246","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-universal-llm-reward-design-through","title":"Boosting Universal LLM Reward Design through the Heuristic Reward Observation Space Evolution","date":"2025-04-10","arxiv_id":"2504.07596","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-day-to-day","title":"Deep Reinforcement Learning for Day-to-day Dynamic Tolling in Tradable Credit Schemes","date":"2025-04-10","arxiv_id":"2504.08074","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-adaptation-with-behavioral-foundation","title":"Fast Adaptation with Behavioral Foundation Models","date":"2025-04-10","arxiv_id":"2504.07896","repositories_listed":0,"syntology":null},{"url":null,"slug":"genetic-programming-with-reinforcement","title":"Genetic Programming with Reinforcement Learning Trained Transformer for Real-World Dynamic Scheduling Problems","date":"2025-04-10","arxiv_id":"2504.07779","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-based-control-of-uas-subject-to","title":"RL-based Control of UAS Subject to Significant Disturbance","date":"2025-04-10","arxiv_id":"2504.08114","repositories_listed":0,"syntology":null},{"url":null,"slug":"better-decisions-through-the-right-causal","title":"Better Decisions through the Right Causal World Model","date":"2025-04-09","arxiv_id":"2504.07257","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-exploration-in-reinforcement-learning","title":"Smart Exploration in Reinforcement Learning using Bounded Uncertainty Models","date":"2025-04-08","arxiv_id":"2504.05978","repositories_listed":0,"syntology":null},{"url":null,"slug":"stratified-expert-cloning-with-adaptive","title":"Stratified Expert Cloning with Adaptive Selection for User Retention in Large-Scale Recommender Systems","date":"2025-04-08","arxiv_id":"2504.05628","repositories_listed":0,"syntology":null},{"url":null,"slug":"tw-crl-time-weighted-contrastive-reward","title":"TW-CRL: Time-Weighted Contrastive Reward Learning for Efficient Inverse Reinforcement Learning","date":"2025-04-08","arxiv_id":"2504.05585","repositories_listed":0,"syntology":null},{"url":null,"slug":"xmtf-a-formula-free-model-for-reinforcement","title":"xMTF: A Formula-Free Model for Reinforcement-Learning-Based Multi-Task Fusion in Recommender Systems","date":"2025-04-08","arxiv_id":"2504.05669","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithm-discovery-with-llms-evolutionary","title":"Algorithm Discovery With LLMs: Evolutionary Search Meets Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.05108","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-informed-modularized-neural-network","title":"Physics-informed Modularized Neural Network for Advanced Building Control by Deep Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.05397","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-role-of-environment-access-in-agnostic","title":"The Role of Environment Access in Agnostic Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.05405","repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-of-price-inflation-on-algorithmic","title":"Impact of Price Inflation on Algorithmic Collusion Through Reinforcement Learning Agents","date":"2025-04-05","arxiv_id":"2504.05335","repositories_listed":0,"syntology":null},{"url":null,"slug":"orbitzoo-multi-agent-reinforcement-learning","title":"OrbitZoo: Multi-Agent Reinforcement Learning Environment for Orbital Dynamics","date":"2025-04-05","arxiv_id":"2504.04160","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithmic-prompt-generation-for-diverse","title":"Algorithmic Prompt Generation for Diverse Human-like Teaming and Communication with Large Language Models","date":"2025-04-04","arxiv_id":"2504.03991","repositories_listed":0,"syntology":null},{"url":null,"slug":"decision-spikeformer-spike-driven-transformer","title":"Decision SpikeFormer: Spike-Driven Transformer for Decision Making","date":"2025-04-04","arxiv_id":"2504.03800","repositories_listed":0,"syntology":null},{"url":null,"slug":"dexterous-manipulation-through-imitation","title":"Dexterous Manipulation through Imitation Learning: A Survey","date":"2025-04-04","arxiv_id":"2504.03515","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-penalty-based-bidirectional","title":"Enhanced Penalty-based Bidirectional Reinforcement Learning Algorithms","date":"2025-04-04","arxiv_id":"2504.03163","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-mixed-criticality-scheduling-with","title":"Improving Mixed-Criticality Scheduling with Reinforcement Learning","date":"2025-04-04","arxiv_id":"2504.03994","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dual-arm-coordination-for-grasping","title":"Learning Dual-Arm Coordination for Grasping Large Flat Objects","date":"2025-04-04","arxiv_id":"2504.03500","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-and-distributional-reinforcement-1","title":"Offline and Distributional Reinforcement Learning for Wireless Communications","date":"2025-04-04","arxiv_id":"2504.03804","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-world-models-with-latent-state","title":"Adapting World Models with Latent-State Dynamics Residuals","date":"2025-04-03","arxiv_id":"2504.02252","repositories_listed":0,"syntology":null},{"url":null,"slug":"inference-time-scaling-for-generalist-reward","title":"Inference-Time Scaling for Generalist Reward Modeling","date":"2025-04-03","arxiv_id":"2504.02495","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-human-knowledge-through-action","title":"Integrating Human Knowledge Through Action Masking in Reinforcement Learning for Operations Research","date":"2025-04-03","arxiv_id":"2504.02662","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-solving-the-1","title":"Reinforcement Learning for Solving the Pricing Problem in Column Generation: Applications to Vehicle Routing","date":"2025-04-03","arxiv_id":"2504.02383","repositories_listed":0,"syntology":null},{"url":null,"slug":"de-novo-molecular-design-enabled-by-direct","title":"De Novo Molecular Design Enabled by Direct Preference Optimization and Curriculum Learning","date":"2025-04-02","arxiv_id":"2504.01389","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-curriculum-learning-for-goal","title":"Probabilistic Curriculum Learning for Goal-Based Reinforcement Learning","date":"2025-04-02","arxiv_id":"2504.01459","repositories_listed":0,"syntology":null},{"url":null,"slug":"grounding-multimodal-llms-to-embodied-agents","title":"Grounding Multimodal LLMs to Embodied Agents that Ask for Help with Reinforcement Learning","date":"2025-04-01","arxiv_id":"2504.00907","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-difficulty-aware-staged-reinforcement","title":"How Difficulty-Aware Staged Reinforcement Learning Enhances LLMs' Reasoning Capabilities: A Preliminary Experimental Study","date":"2025-04-01","arxiv_id":"2504.00829","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-reinforcement-learning-based","title":"A Survey of Reinforcement Learning-Based Motion Planning for Autonomous Driving: Lessons Learned from a Driving Task Perspective","date":"2025-03-31","arxiv_id":"2503.23650","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-high-efficiency-organic","title":"Accelerating High-Efficiency Organic Photovoltaic Discovery via Pretrained Graph Neural Networks and Generative Reinforcement Learning","date":"2025-03-31","arxiv_id":"2503.23766","repositories_listed":0,"syntology":null},{"url":null,"slug":"fair-dynamic-spectrum-access-via-fully","title":"Fair Dynamic Spectrum Access via Fully Decentralized Multi-Agent Reinforcement Learning","date":"2025-03-31","arxiv_id":"2503.24296","repositories_listed":0,"syntology":null},{"url":null,"slug":"hacts-a-human-as-copilot-teleoperation-system","title":"HACTS: a Human-As-Copilot Teleoperation System for Robot Learning","date":"2025-03-31","arxiv_id":"2503.24070","repositories_listed":0,"syntology":null},{"url":null,"slug":"judgelrm-large-reasoning-models-as-a-judge","title":"JudgeLRM: Large Reasoning Models as a Judge","date":"2025-03-31","arxiv_id":"2504.00050","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-based-reward-modulated-learning","title":"Noise-based reward-modulated learning","date":"2025-03-31","arxiv_id":"2503.23972","repositories_listed":0,"syntology":null},{"url":null,"slug":"nuclear-microreactor-control-with-deep","title":"Nuclear Microreactor Control with Deep Reinforcement Learning","date":"2025-03-31","arxiv_id":"2504.00156","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-safe-autonomous","title":"Reinforcement Learning for Safe Autonomous Two Device Navigation of Cerebral Vessels in Mechanical Thrombectomy","date":"2025-03-31","arxiv_id":"2503.24140","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-systematic-decade-review-of-trip-route","title":"A Systematic Decade Review of Trip Route Planning with Travel Time Estimation based on User Preferences and Behavior","date":"2025-03-30","arxiv_id":"2503.23486","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-deep-learning-and-large-language","title":"Advanced Deep Learning and Large Language Models: Comprehensive Insights for Cancer Detection","date":"2025-03-30","arxiv_id":"2504.13186","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-active-matter","title":"Reinforcement Learning for Active Matter","date":"2025-03-30","arxiv_id":"2503.23308","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-graph","title":"Multi-Agent Reinforcement Learning for Graph Discovery in D2D-Enabled Federated Learning","date":"2025-03-29","arxiv_id":"2503.23218","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-sql-reinforcement-learning-with-sql","title":"Reasoning-SQL: Reinforcement Learning with SQL Tailored Partial Rewards for Reasoning-Enhanced Text-to-SQL","date":"2025-03-29","arxiv_id":"2503.23157","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl2grid-benchmarking-reinforcement-learning","title":"RL2Grid: Benchmarking Reinforcement Learning in Power Grid Operations","date":"2025-03-29","arxiv_id":"2503.23101","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-guided-sequence-weighting-for","title":"Entropy-guided sequence weighting for efficient exploration in RL-based LLM fine-tuning","date":"2025-03-28","arxiv_id":"2503.22456","repositories_listed":0,"syntology":null},{"url":null,"slug":"flam-foundation-model-based-body","title":"FLAM: Foundation Model-Based Body Stabilization for Humanoid Locomotion and Manipulation","date":"2025-03-28","arxiv_id":"2503.22249","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-machine-learning","title":"Reinforcement Learning for Machine Learning Model Deployment: Evaluating Multi-Armed Bandits in ML Ops Environments","date":"2025-03-28","arxiv_id":"2503.22595","repositories_listed":0,"syntology":null},{"url":null,"slug":"bresa-bio-inspired-reflexive-safe","title":"Bresa: Bio-inspired Reflexive Safe Reinforcement Learning for Contact-Rich Robotic Tasks","date":"2025-03-27","arxiv_id":"2503.21989","repositories_listed":0,"syntology":null},{"url":null,"slug":"harmonia-a-multi-agent-reinforcement-learning","title":"Harmonia: A Multi-Agent Reinforcement Learning Approach to Data Placement and Migration in Hybrid Storage Systems","date":"2025-03-26","arxiv_id":"2503.20507","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-adaptive-dexterous-grasping-from","title":"Learning Adaptive Dexterous Grasping from Single Demonstrations","date":"2025-03-26","arxiv_id":"2503.20208","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-offline-reinforcement-learning-3","title":"Model-Based Offline Reinforcement Learning with Adversarial Data Augmentation","date":"2025-03-26","arxiv_id":"2503.20285","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-discrete","title":"Offline Reinforcement Learning with Discrete Diffusion Skills","date":"2025-03-26","arxiv_id":"2503.20176","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-beyond-limits-advances-and-open","title":"Reasoning Beyond Limits: Advances and Open Problems for LLMs","date":"2025-03-26","arxiv_id":"2503.22732","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-world-models-for-bilevel","title":"Synthesizing world models for bilevel planning","date":"2025-03-26","arxiv_id":"2503.20124","repositories_listed":0,"syntology":null},{"url":null,"slug":"tar-teacher-aligned-representations-via","title":"TAR: Teacher-Aligned Representations via Contrastive Learning for Quadrupedal Locomotion","date":"2025-03-26","arxiv_id":"2503.20839","repositories_listed":0,"syntology":null}],"record_sha256":"cc09ed026c9127d8332997276beb6249949108059053640c3a80c69b3f52622a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}