{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/56","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":56,"pages_in_order":152,"rows_per_page":100,"rows":[5501,5600],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/55","next":"/task/reinforcement-learning-1/papers/57","papers":[{"url":null,"slug":"dynamic-optimization-of-storage-systems-using","title":"Dynamic Optimization of Storage Systems Using Reinforcement Learning Techniques","date":"2024-12-29","arxiv_id":"2501.00068","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-code-llms-with-reinforcement","title":"Enhancing Code LLMs with Reinforcement Learning in Code Generation: A Survey","date":"2024-12-29","arxiv_id":"2412.20367","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-conditioned-data-augmentation-for","title":"Goal-Conditioned Data Augmentation for Offline Reinforcement Learning","date":"2024-12-29","arxiv_id":"2412.20519","repositories_listed":0,"syntology":null},{"url":null,"slug":"election-of-collaborators-via-reinforcement","title":"Election of Collaborators via Reinforcement Learning for Federated Brain Tumor Segmentation","date":"2024-12-28","arxiv_id":"2412.20253","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-attention-based-casual-discovery-with","title":"Graph-attention-based Casual Discovery with Trust Region-navigated Clipping Policy Optimization","date":"2024-12-27","arxiv_id":"2412.19578","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-task-mapping","title":"A Reinforcement Learning-Based Task Mapping Method to Improve the Reliability of Clustered Manycores","date":"2024-12-26","arxiv_id":"2412.19340","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimal-batch-adaptive-learning-policy-engine","title":"Minimal Batch Adaptive Learning Policy Engine for Real-Time Mid-Price Forecasting in High-Frequency Trading","date":"2024-12-26","arxiv_id":"2412.19372","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-fantasy-sports-team-selection-with","title":"Optimizing Fantasy Sports Team Selection with Deep Reinforcement Learning","date":"2024-12-26","arxiv_id":"2412.19215","repositories_listed":0,"syntology":null},{"url":null,"slug":"preventive-energy-management-for-distribution","title":"Preventive Energy Management for Distribution Systems Under Uncertain Events: A Deep Reinforcement Learning Approach","date":"2024-12-26","arxiv_id":"2412.19382","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-exploration-in-reward","title":"Provably Efficient Exploration in Reward Machines with Low Regret","date":"2024-12-26","arxiv_id":"2412.19194","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-multi-step-reasoning-abilities-of","title":"Improving Multi-Step Reasoning Abilities of Large Language Models with Direct Advantage Policy Optimization","date":"2024-12-24","arxiv_id":"2412.18279","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-framework-for-reinforcement-learning","title":"Quantum framework for Reinforcement Learning: Integrating Markov decision process, quantum arithmetic, and trajectory search","date":"2024-12-24","arxiv_id":"2412.18208","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-deep-reinforcement-learning-for","title":"Multimodal Deep Reinforcement Learning for Portfolio Optimization","date":"2024-12-23","arxiv_id":"2412.17293","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-prompt-strategies-for-the-segment","title":"Optimizing Prompt Strategies for SAM: Advancing lesion Segmentation Across Diverse Medical Imaging Modalities","date":"2024-12-23","arxiv_id":"2412.17943","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-motor-control-a","title":"Reinforcement Learning for Motor Control: A Comprehensive Review","date":"2024-12-23","arxiv_id":"2412.17936","repositories_listed":0,"syntology":null},{"url":null,"slug":"acl-ql-adaptive-conservative-level-in-q","title":"ACL-QL: Adaptive Conservative Level in Q-Learning for Offline Reinforcement Learning","date":"2024-12-22","arxiv_id":"2412.16848","repositories_listed":0,"syntology":null},{"url":null,"slug":"adam-on-local-time-addressing-nonstationarity","title":"Adam on Local Time: Addressing Nonstationarity in RL with Relative Adam Timesteps","date":"2024-12-22","arxiv_id":"2412.17113","repositories_listed":0,"syntology":null},{"url":null,"slug":"environment-descriptions-for-usability-and","title":"Environment Descriptions for Usability and Generalisation in Reinforcement Learning","date":"2024-12-22","arxiv_id":"2412.16970","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathematics-and-machine-creativity-a-survey","title":"Mathematics and Machine Creativity: A Survey on Bridging Mathematics with AI","date":"2024-12-21","arxiv_id":"2412.16543","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-enhancing-network-throughput-using","title":"On Enhancing Network Throughput using Reinforcement Learning in Sliced Testbeds","date":"2024-12-21","arxiv_id":"2412.16673","repositories_listed":0,"syntology":null},{"url":null,"slug":"subgoal-discovery-using-a-free-energy","title":"Subgoal Discovery Using a Free Energy Paradigm and State Aggregations","date":"2024-12-21","arxiv_id":"2412.16687","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-general-to-specific-tailoring-large","title":"From General to Specific: Tailoring Large Language Models for Personalized Healthcare","date":"2024-12-20","arxiv_id":"2412.15957","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-low-speed-autonomous-driving-a","title":"Optimizing Low-Speed Autonomous Driving: A Reinforcement Learning Approach to Route Stability and Maximum Speed","date":"2024-12-20","arxiv_id":"2412.16248","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlm-rl-a-unified-vision-language-models-and","title":"VLM-RL: A Unified Vision Language Models and Reinforcement Learning Framework for Safe Autonomous Driving","date":"2024-12-20","arxiv_id":"2412.15544","repositories_listed":0,"syntology":null},{"url":null,"slug":"adacred-adaptive-causal-decision-transformers","title":"AdaCred: Adaptive Causal Decision Transformers with Feature Crediting","date":"2024-12-19","arxiv_id":"2412.15427","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-generate-research-idea-with","title":"Learning to Generate Research Idea with Dynamic Control","date":"2024-12-19","arxiv_id":"2412.14626","repositories_listed":0,"syntology":null},{"url":null,"slug":"simulation-free-hierarchical-latent-policy","title":"Simulation-Free Hierarchical Latent Policy Planning for Proactive Dialogues","date":"2024-12-19","arxiv_id":"2412.14584","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-loop-federated-actor-critic-across","title":"Single-Loop Federated Actor-Critic across Heterogeneous Environments","date":"2024-12-19","arxiv_id":"2412.14555","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-critique-tune-based-reinforcement","title":"Bayesian Critique-Tune-Based Reinforcement Learning with Adaptive Pressure for Multi-Intersection Traffic Signal Control","date":"2024-12-18","arxiv_id":"2412.16225","repositories_listed":0,"syntology":null},{"url":null,"slug":"harvesting-energy-from-turbulent-winds-with","title":"Harvesting energy from turbulent winds with Reinforcement Learning","date":"2024-12-18","arxiv_id":"2412.13961","repositories_listed":0,"syntology":null},{"url":null,"slug":"inference-aware-fine-tuning-for-best-of-n","title":"Inference-Aware Fine-Tuning for Best-of-N Sampling in Large Language Models","date":"2024-12-18","arxiv_id":"2412.15287","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-rldrive-human-aligned-autonomous-driving","title":"CLIP-RLDrive: Human-Aligned Autonomous Driving via CLIP-Based Reward Shaping in Reinforcement Learning","date":"2024-12-17","arxiv_id":"2412.16201","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-of-restricted-normalizing-flow-towards","title":"Design of Restricted Normalizing Flow towards Arbitrary Stochastic Policy with Computational Efficiency","date":"2024-12-17","arxiv_id":"2412.12894","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-visuotactile-estimation-and-control","title":"Learning Visuotactile Estimation and Control for Non-prehensile Manipulation under Occlusions","date":"2024-12-17","arxiv_id":"2412.13157","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-reinforcement-learning-for","title":"Multi-Task Reinforcement Learning for Quadrotors","date":"2024-12-17","arxiv_id":"2412.12442","repositories_listed":0,"syntology":null},{"url":null,"slug":"parmod-a-parallel-and-modular-framework-for","title":"ParMod: A Parallel and Modular Framework for Learning Non-Markovian Tasks","date":"2024-12-17","arxiv_id":"2412.12700","repositories_listed":0,"syntology":null},{"url":"/paper/efficient-policy-adaptation-with-contrastive-1","slug":"efficient-policy-adaptation-with-contrastive-1","title":"Efficient Policy Adaptation with Contrastive Prompt Ensemble for Embodied Agents","date":"2024-12-16","arxiv_id":"2412.11484","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/efficient-policy-adaptation-with-contrastive-1#ran","syntology_url":"https://syntology.ai/paper/2412.11484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11484"}},"official":null}},{"url":null,"slug":"equivariant-action-sampling-for-reinforcement","title":"Equivariant Action Sampling for Reinforcement Learning and Planning","date":"2024-12-16","arxiv_id":"2412.12237","repositories_listed":0,"syntology":null},{"url":null,"slug":"maxinforl-boosting-exploration-in","title":"MaxInfoRL: Boosting exploration in reinforcement learning through information gain maximization","date":"2024-12-16","arxiv_id":"2412.12098","repositories_listed":0,"syntology":null},{"url":null,"slug":"mgda-model-based-goal-data-augmentation-for","title":"MGDA: Model-based Goal Data Augmentation for Offline Goal-conditioned Weighted Supervised Learning","date":"2024-12-16","arxiv_id":"2412.11410","repositories_listed":0,"syntology":null},{"url":null,"slug":"stabilizing-reinforcement-learning-in","title":"Stabilizing Reinforcement Learning in Differentiable Multiphysics Simulation","date":"2024-12-16","arxiv_id":"2412.12089","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-driving-with-evolution-capability-a","title":"Automated Driving with Evolution Capability: A Reinforcement Learning Method with Monotonic Performance Enhancement","date":"2024-12-14","arxiv_id":"2412.10822","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-time-optimal-investment-with","title":"Continuous-time optimal investment with portfolio constraints: a reinforcement learning approach","date":"2024-12-14","arxiv_id":"2412.10692","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-scalable","title":"Deep Reinforcement Learning for Scalable Multiagent Spacecraft Inspection","date":"2024-12-13","arxiv_id":"2412.10530","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-instrument-design-with-reinforcement","title":"Physics Instrument Design with Reinforcement Learning","date":"2024-12-13","arxiv_id":"2412.10237","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-machine-inference-for-robotic","title":"Reward Machine Inference for Robotic Manipulation","date":"2024-12-13","arxiv_id":"2412.10096","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-text-to-trajectory-exploring-complex","title":"From Text to Trajectory: Exploring Complex Constraint Representation and Decomposition in Safe Reinforcement Learning","date":"2024-12-12","arxiv_id":"2412.08920","repositories_listed":0,"syntology":null},{"url":null,"slug":"pickllm-context-aware-rl-assisted-large","title":"PickLLM: Context-Aware RL-Assisted Large Language Model Routing","date":"2024-12-12","arxiv_id":"2412.12170","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-train-based-distributed-multi-agent","title":"Quantum-Train-Based Distributed Multi-Agent Reinforcement Learning","date":"2024-12-12","arxiv_id":"2412.08845","repositories_listed":0,"syntology":null},{"url":null,"slug":"radiology-report-generation-via-multi","title":"Radiology Report Generation via Multi-objective Preference Optimization","date":"2024-12-12","arxiv_id":"2412.08901","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-within-the-classical","title":"Reinforcement Learning Within the Classical Robotics Stack: A Case Study in Robot Soccer","date":"2024-12-12","arxiv_id":"2412.09417","repositories_listed":0,"syntology":null},{"url":null,"slug":"ask1-development-and-reinforcement-learning","title":"Ask1: Development and Reinforcement Learning-Based Control of a Custom Quadruped Robot","date":"2024-12-11","arxiv_id":"2412.08019","repositories_listed":0,"syntology":null},{"url":null,"slug":"coarse-to-fine-a-dual-phase-channel-adaptive","title":"Coarse-to-Fine: A Dual-Phase Channel-Adaptive Method for Wireless Image Transmission","date":"2024-12-11","arxiv_id":"2412.08211","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobile-television-predictive-motion-priors","title":"Mobile-TeleVision: Predictive Motion Priors for Humanoid Whole-Body Control","date":"2024-12-10","arxiv_id":"2412.07773","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-sensor-redundancy-in-sequential","title":"Optimizing Sensor Redundancy in Sequential Decision-Making Problems","date":"2024-12-10","arxiv_id":"2412.07686","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-and-sequential-text-to-image","title":"Preference Adaptive and Sequential Text-to-Image Generation","date":"2024-12-10","arxiv_id":"2412.10419","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-resolution-policy-distillation","title":"Progressive-Resolution Policy Distillation: Leveraging Coarse-Resolution Simulations for Time-Efficient Fine-Resolution Policy Learning","date":"2024-12-10","arxiv_id":"2412.07477","repositories_listed":0,"syntology":null},{"url":null,"slug":"swarm-behavior-cloning","title":"Swarm Behavior Cloning","date":"2024-12-10","arxiv_id":"2412.07617","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-agnostic-rl-offline-rl-and-online-rl","title":"Policy Agnostic RL: Offline RL and Online RL Fine-Tuning of Any Class and Backbone","date":"2024-12-09","arxiv_id":"2412.06685","repositories_listed":0,"syntology":null},{"url":null,"slug":"skill-enhanced-reinforcement-learning","title":"Skill-Enhanced Reinforcement Learning Acceleration from Demonstrations","date":"2024-12-09","arxiv_id":"2412.06207","repositories_listed":0,"syntology":null},{"url":null,"slug":"unraveling-the-complexity-of-memory-in-rl","title":"Unraveling the Complexity of Memory in RL Agents: an Approach for Classification and Evaluation","date":"2024-12-09","arxiv_id":"2412.06531","repositories_listed":0,"syntology":null},{"url":null,"slug":"mean-variance-portfolio-selection-by","title":"Mean--Variance Portfolio Selection by Continuous-Time Reinforcement Learning: Algorithms, Regret Analysis, and Empirical Study","date":"2024-12-08","arxiv_id":"2412.16175","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-a-discrete-time","title":"Reinforcement Learning for a Discrete-Time Linear-Quadratic Control Problem with an Application","date":"2024-12-08","arxiv_id":"2412.05906","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-planning-a-primer-and-survey-preliminary","title":"AI Planning: A Primer and Survey (Preliminary Report)","date":"2024-12-07","arxiv_id":"2412.05528","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-soft-driving-constraints-from","title":"Learning Soft Driving Constraints from Vectorized Scene Embeddings while Imitating Expert Trajectories","date":"2024-12-07","arxiv_id":"2412.05717","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-zero-zero-shot-language-to-behaviors","title":"RLZero: Direct Policy Inference from Language Without In-Domain Supervision","date":"2024-12-07","arxiv_id":"2412.05718","repositories_listed":0,"syntology":null},{"url":null,"slug":"element-episodic-and-lifelong-exploration-via","title":"ELEMENT: Episodic and Lifelong Exploration via Maximum Entropy","date":"2024-12-05","arxiv_id":"2412.03800","repositories_listed":0,"syntology":null},{"url":null,"slug":"finer-behavioral-foundation-models-via-auto","title":"Finer Behavioral Foundation Models via Auto-Regressive Features and Advantage Weighting","date":"2024-12-05","arxiv_id":"2412.04368","repositories_listed":0,"syntology":null},{"url":null,"slug":"traffic-co-simulation-framework-empowered-by","title":"Traffic Co-Simulation Framework Empowered by Infrastructure Camera Sensing and Reinforcement Learning","date":"2024-12-05","arxiv_id":"2412.03925","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyper-hyperparameter-robust-efficient","title":"Hyper: Hyperparameter Robust Efficient Exploration in Reinforcement Learning","date":"2024-12-04","arxiv_id":"2412.03767","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-whole-body-loco-manipulation-for","title":"Learning Whole-Body Loco-Manipulation for Omni-Directional Task Space Pose Tracking with a Wheeled-Quadrupedal-Manipulator","date":"2024-12-04","arxiv_id":"2412.03012","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-deep-reinforcement-learning-to-enhance","title":"Using Deep Reinforcement Learning to Enhance Channel Sampling Patterns in Integrated Sensing and Communication","date":"2024-12-04","arxiv_id":"2412.03157","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-driven-resource-allocation-framework-for","title":"AI-Driven Resource Allocation Framework for Microservices in Hybrid Cloud Platforms","date":"2024-12-03","arxiv_id":"2412.02610","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-critical-scenarios-for-testing","title":"Generating Critical Scenarios for Testing Automated Driving Systems","date":"2024-12-03","arxiv_id":"2412.02574","repositories_listed":0,"syntology":null},{"url":null,"slug":"out-of-distribution-detection-for","title":"Out-of-Distribution Detection for Neurosymbolic Autonomous Cyber Agents","date":"2024-12-03","arxiv_id":"2412.02875","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-reviews-of-bandit-problems-in-ai","title":"Selective Reviews of Bandit Problems in AI via a Statistical View","date":"2024-12-03","arxiv_id":"2412.02251","repositories_listed":0,"syntology":null},{"url":null,"slug":"technical-report-on-reinforcement-learning","title":"Technical Report on Reinforcement Learning Control on the Lucas-Nülle Inverted Pendulum","date":"2024-12-03","arxiv_id":"2412.02264","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-memory-based-reinforcement-learning","title":"A Memory-Based Reinforcement Learning Approach to Integrated Sensing and Communication","date":"2024-12-02","arxiv_id":"2412.01077","repositories_listed":0,"syntology":null},{"url":null,"slug":"dense-dynamics-aware-reward-synthesis","title":"Dense Dynamics-Aware Reward Synthesis: Integrating Prior Experience with Demonstrations","date":"2024-12-02","arxiv_id":"2412.01114","repositories_listed":0,"syntology":null},{"url":null,"slug":"explore-reinforced-equilibrium-approximation","title":"Explore Reinforced: Equilibrium Approximation with Reinforcement Learning","date":"2024-12-02","arxiv_id":"2412.02016","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl2-reinforce-large-language-model-to-assist","title":"RL2: Reinforce Large Language Model to Assist Safe Reinforcement Learning for Energy Management of Active Distribution Networks","date":"2024-12-02","arxiv_id":"2412.01303","repositories_listed":0,"syntology":null},{"url":null,"slug":"bilinear-convolution-decomposition-for-causal","title":"Bilinear Convolution Decomposition for Causal RL Interpretability","date":"2024-12-01","arxiv_id":"2412.00944","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-partially-observable-reinforcement","title":"Provable Partially Observable Reinforcement Learning with Privileged Information","date":"2024-12-01","arxiv_id":"2412.00985","repositories_listed":0,"syntology":null},{"url":"/paper/bots-batch-bayesian-optimization-of-extended","slug":"bots-batch-bayesian-optimization-of-extended","title":"BOTS: Batch Bayesian Optimization of Extended Thompson Sampling for Severely Episode-Limited RL Settings","date":"2024-11-30","arxiv_id":"2412.00308","repositories_listed":0,"syntology":{"n":9,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/bots-batch-bayesian-optimization-of-extended#ran","syntology_url":"https://syntology.ai/paper/2412.00308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.00308"}},"official":null}},{"url":null,"slug":"hvac-dpt-a-decision-pretrained-transformer","title":"HVAC-DPT: A Decision Pretrained Transformer for HVAC Control","date":"2024-11-29","arxiv_id":"2411.19746","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-milp-solver-a-reinforcement-learning","title":"RL-MILP Solver: A Reinforcement Learning Approach for Solving Mixed-Integer Linear Programs with Graph Neural Networks","date":"2024-11-29","arxiv_id":"2411.19517","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-rubik-s-cube-without-tricky-sampling","title":"Solving Rubik's Cube Without Tricky Sampling","date":"2024-11-29","arxiv_id":"2411.19583","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-survey-of-reinforcement","title":"A Comprehensive Survey of Reinforcement Learning: From Algorithms to Practical Challenges","date":"2024-11-28","arxiv_id":"2411.18892","repositories_listed":0,"syntology":null},{"url":null,"slug":"convex-regularization-and-convergence-of","title":"Convex Regularization and Convergence of Policy Gradient Flows under Safety Constraints","date":"2024-11-28","arxiv_id":"2411.19193","repositories_listed":0,"syntology":null},{"url":null,"slug":"tea-trajectory-encoding-augmentation-for","title":"TEA: Trajectory Encoding Augmentation for Robust and Transferable Policies in Offline Reinforcement Learning","date":"2024-11-28","arxiv_id":"2411.19133","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-non-prehensile-object-transport-via","title":"Dynamic Non-Prehensile Object Transport via Model-Predictive Reinforcement Learning","date":"2024-11-27","arxiv_id":"2412.00086","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-retail-pricing-via-q-learning-a","title":"Dynamic Retail Pricing via Q-Learning -- A Reinforcement Learning Framework for Enhanced Revenue Management","date":"2024-11-27","arxiv_id":"2411.18261","repositories_listed":0,"syntology":null},{"url":null,"slug":"elemental-interactive-learning-from","title":"ELEMENTAL: Interactive Learning from Demonstrations and Vision-Language Models for Reward Design in Robotics","date":"2024-11-27","arxiv_id":"2411.18825","repositories_listed":0,"syntology":null},{"url":null,"slug":"neohebbian-synapses-to-accelerate-online","title":"NeoHebbian Synapses to Accelerate Online Training of Neuromorphic Hardware","date":"2024-11-27","arxiv_id":"2411.18272","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaleviz-scaling-visualization-recommendation","title":"ScaleViz: Scaling Visualization Recommendation Models on Large Data","date":"2024-11-27","arxiv_id":"2411.18657","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-proximal-policy-optimization","title":"Accelerating Proximal Policy Optimization Learning Using Task Prediction for Solving Environments with Delayed Rewards","date":"2024-11-26","arxiv_id":"2411.17861","repositories_listed":0,"syntology":null},{"url":null,"slug":"free-2-guide-gradient-free-path-integral","title":"Free$^2$Guide: Gradient-Free Path Integral Control for Enhancing Text-to-Video Generation with Large Vision-Language Models","date":"2024-11-26","arxiv_id":"2411.17041","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-based-offline-learning-for-embodied","title":"LLM-Based Offline Learning for Embodied Agents via Consistency-Guided Reward Ensemble","date":"2024-11-26","arxiv_id":"2411.17135","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressor-a-perceptually-guided-reward","title":"PROGRESSOR: A Perceptually Guided Reward Estimator with Self-Supervised Online Refinement","date":"2024-11-26","arxiv_id":"2411.17764","repositories_listed":0,"syntology":null},{"url":null,"slug":"m3-mamba-assisted-multi-circuit-optimization","title":"M3: Mamba-assisted Multi-Circuit Optimization via MBRL with Effective Scheduling","date":"2024-11-25","arxiv_id":"2411.16019","repositories_listed":0,"syntology":null}],"record_sha256":"e6869bcfbfa4d3ea3d7b85b86008032823ada916f7f123b8b9d2e6f23d1d5e99","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}