{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/46","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":46,"pages_in_order":135,"rows_per_page":100,"rows":[4501,4600],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/45","next":"/task/reinforcement-learning-2/papers/47","papers":[{"url":null,"slug":"mpo-an-efficient-post-processing-framework","title":"MPO: An Efficient Post-Processing Framework for Mixing Diverse Preference Alignment","date":"2025-02-25","arxiv_id":"2502.18699","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-on-strategic-mining-in-blockchain-a","title":"Survey on Strategic Mining in Blockchain: A Reinforcement Learning Approach","date":"2025-02-24","arxiv_id":"2502.17307","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-reinforcement-learning-for","title":"Towards Reinforcement Learning for Exploration of Speculative Execution Vulnerabilities","date":"2025-02-24","arxiv_id":"2502.16756","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-sentiment-manipulation-by-llm","title":"Exploring Sentiment Manipulation by LLM-Enabled Intelligent Trading Agents","date":"2025-02-22","arxiv_id":"2502.16343","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-user-level-private-reinforcement","title":"Towards User-level Private Reinforcement Learning with Human Feedback","date":"2025-02-22","arxiv_id":"2502.17515","repositories_listed":0,"syntology":null},{"url":"/paper/hyperspherical-normalization-for-scalable","slug":"hyperspherical-normalization-for-scalable","title":"Hyperspherical Normalization for Scalable Deep Reinforcement Learning","date":"2025-02-21","arxiv_id":"2502.15280","repositories_listed":0,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hyperspherical-normalization-for-scalable#ran","syntology_url":"https://syntology.ai/paper/2502.15280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15280"}},"official":null}},{"url":null,"slug":"the-evolving-landscape-of-llm-and-vlm","title":"The Evolving Landscape of LLM- and VLM-Integrated Reinforcement Learning","date":"2025-02-21","arxiv_id":"2502.15214","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-reward-free-reinforcement-learning","title":"Towards a Reward-Free Reinforcement Learning Framework for Vehicle Control","date":"2025-02-21","arxiv_id":"2502.15262","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-mean-field-multi-agent-reinforcement","title":"Causal Mean Field Multi-Agent Reinforcement Learning","date":"2025-02-20","arxiv_id":"2502.14200","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-q-learning-an-ill-posed-problem","title":"Is Q-learning an Ill-posed Problem?","date":"2025-02-20","arxiv_id":"2502.14365","repositories_listed":0,"syntology":null},{"url":null,"slug":"mrl-discovering-transient-execution","title":"μRL: Discovering Transient Execution Vulnerabilities Using Reinforcement Learning","date":"2025-02-20","arxiv_id":"2502.14307","repositories_listed":0,"syntology":null},{"url":null,"slug":"sprig-stackelberg-perception-reinforcement","title":"SPRIG: Stackelberg Perception-Reinforcement Learning with Internal Game Dynamics","date":"2025-02-20","arxiv_id":"2502.14264","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-target-radar-search-and-track-using","title":"Multi-Target Radar Search and Track Using Sequence-Capable Deep Reinforcement Learning","date":"2025-02-19","arxiv_id":"2502.13584","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-quantification-for-markov-chains","title":"Uncertainty quantification for Markov chains with application to temporal difference learning","date":"2025-02-19","arxiv_id":"2502.13822","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-learning-conversational-ai-a","title":"Continuous Learning Conversational AI: A Personalized Agent Framework via A2C Reinforcement Learning","date":"2025-02-18","arxiv_id":"2502.12876","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-repair-with-reinforcement-learning","title":"Implicit Repair with Reinforcement Learning in Emergent Communication","date":"2025-02-18","arxiv_id":"2502.12624","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-transformers-as-iterative","title":"Self-Supervised Transformers as Iterative Solution Improvers for Constraint Satisfaction","date":"2025-02-18","arxiv_id":"2502.15794","repositories_listed":0,"syntology":null},{"url":null,"slug":"theorem-prover-as-a-judge-for-synthetic-data","title":"Theorem Prover as a Judge for Synthetic Data Generation","date":"2025-02-18","arxiv_id":"2502.13137","repositories_listed":0,"syntology":null},{"url":null,"slug":"fitlight-federated-imitation-learning-for","title":"FitLight: Federated Imitation Learning for Plug-and-Play Autonomous Traffic Signal Control","date":"2025-02-17","arxiv_id":"2502.11937","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-mobile-ai-generated-content","title":"Intelligent Mobile AI-Generated Content Services via Interactive Prompt Engineering and Dynamic Service Provisioning","date":"2025-02-17","arxiv_id":"2502.11386","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-reason-at-the-frontier-of","title":"Learning to Reason at the Frontier of Learnability","date":"2025-02-17","arxiv_id":"2502.12272","repositories_listed":0,"syntology":null},{"url":null,"slug":"theoretical-barriers-in-bellman-based","title":"Theoretical Barriers in Bellman-Based Reinforcement Learning","date":"2025-02-17","arxiv_id":"2502.11968","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-online-resource-constrained","title":"Solving Online Resource-Constrained Scheduling for Follow-Up Observation in Astronomy: a Reinforcement Learning Approach","date":"2025-02-16","arxiv_id":"2502.11134","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-tutorial-on-llm-reasoning-relevant-methods","title":"A Tutorial on LLM Reasoning: Relevant Methods behind ChatGPT o1","date":"2025-02-15","arxiv_id":"2502.10867","repositories_listed":0,"syntology":null},{"url":null,"slug":"rule-bottleneck-reinforcement-learning-joint","title":"Rule-Bottleneck Reinforcement Learning: Joint Explanation and Decision Optimization for Resource Allocation with Language Agents","date":"2025-02-15","arxiv_id":"2502.10732","repositories_listed":0,"syntology":null},{"url":null,"slug":"tackling-the-zero-shot-reinforcement-learning","title":"Tackling the Zero-Shot Reinforcement Learning Loss Directly","date":"2025-02-15","arxiv_id":"2502.10792","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-information-prioritization-for","title":"Causal Information Prioritization for Efficient Reinforcement Learning","date":"2025-02-14","arxiv_id":"2502.10097","repositories_listed":0,"syntology":null},{"url":null,"slug":"combinatorial-reinforcement-learning-with","title":"Combinatorial Reinforcement Learning with Preference Feedback","date":"2025-02-14","arxiv_id":"2502.10158","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-we-need-to-verify-step-by-step-rethinking","title":"Do We Need to Verify Step by Step? Rethinking Process Supervision from a Theoretical Perspective","date":"2025-02-14","arxiv_id":"2502.10581","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-reinforcement-learning-for-actors","title":"Dynamic Reinforcement Learning for Actors","date":"2025-02-14","arxiv_id":"2502.10200","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-constrained","title":"Reinforcement Learning based Constrained Optimal Control: an Interpretable Reward Design","date":"2025-02-14","arxiv_id":"2502.10187","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-strategy-based-and","title":"Reinforcement Learning in Strategy-Based and Atari Games: A Review of Google DeepMinds Innovations","date":"2025-02-14","arxiv_id":"2502.10303","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-consistent-model-based-adaptation-for","title":"Self-Consistent Model-based Adaptation for Visual Reinforcement Learning","date":"2025-02-14","arxiv_id":"2502.09923","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-reinforcement-learning-for","title":"A Survey of Reinforcement Learning for Optimization in Automation","date":"2025-02-13","arxiv_id":"2502.09417","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-off-policy-n-step-td-learning","title":"Analysis of Off-Policy $n$-Step TD-Learning with Linear Function Approximation","date":"2025-02-13","arxiv_id":"2502.08941","repositories_listed":0,"syntology":null},{"url":null,"slug":"coupled-rendezvous-and-docking-maneuver","title":"Coupled Rendezvous and Docking Maneuver control of satellite using Reinforcement learning-based Adaptive Fixed-Time Sliding Mode Controller","date":"2025-02-13","arxiv_id":"2502.09517","repositories_listed":0,"syntology":null},{"url":null,"slug":"variable-stiffness-for-robust-locomotion","title":"Variable Stiffness for Robust Locomotion through Reinforcement Learning","date":"2025-02-13","arxiv_id":"2502.09436","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-user","title":"Deep Reinforcement Learning-Based User Scheduling for Collaborative Perception","date":"2025-02-12","arxiv_id":"2502.10456","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-robust-federated-reinforcement","title":"Provably Robust Federated Reinforcement Learning","date":"2025-02-12","arxiv_id":"2502.08123","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-in-context-reinforcement-learning","title":"A Survey of In-Context Reinforcement Learning","date":"2025-02-11","arxiv_id":"2502.07978","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploratory-diffusion-policy-for-unsupervised","title":"Exploratory Diffusion Model for Unsupervised Reinforcement Learning","date":"2025-02-11","arxiv_id":"2502.07279","repositories_listed":0,"syntology":null},{"url":null,"slug":"logarithmic-regret-for-online-kl-regularized","title":"Logarithmic Regret for Online KL-Regularized Reinforcement Learning","date":"2025-02-11","arxiv_id":"2502.07460","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-actuator-attacks-on-autonomous","title":"Optimal Actuator Attacks on Autonomous Vehicles Using Reinforcement Learning","date":"2025-02-11","arxiv_id":"2502.07839","repositories_listed":0,"syntology":null},{"url":null,"slug":"picts-a-novel-deep-reinforcement-learning","title":"PICTS: A Novel Deep Reinforcement Learning Approach for Dynamic P-I Control in Scanning Probe Microscopy","date":"2025-02-11","arxiv_id":"2502.07326","repositories_listed":0,"syntology":null},{"url":null,"slug":"polynomial-time-approximability-of","title":"Polynomial-Time Approximability of Constrained Reinforcement Learning","date":"2025-02-11","arxiv_id":"2502.07764","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-off-policy-reinforcement-learning","title":"Scaling Off-Policy Reinforcement Learning with Batch and Weight Normalization","date":"2025-02-11","arxiv_id":"2502.07523","repositories_listed":0,"syntology":null},{"url":null,"slug":"vsc-rl-advancing-autonomous-vision-language","title":"Advancing Autonomous VLM Agents via Variational Subgoal-Conditioned Reinforcement Learning","date":"2025-02-11","arxiv_id":"2502.07949","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-offloading-in-vehicular-edge-computing","title":"Intelligent Offloading in Vehicular Edge Computing: A Comprehensive Review of Deep Reinforcement Learning Approaches and Architectures","date":"2025-02-10","arxiv_id":"2502.06963","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-explainable-deep-reinforcement","title":"A Survey on Explainable Deep Reinforcement Learning","date":"2025-02-08","arxiv_id":"2502.06869","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-control-of-tandem-wing-experimental","title":"Real Time Control of Tandem-Wing Experimental Platform Using Concerto Reinforcement Learning","date":"2025-02-08","arxiv_id":"2502.10429","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-stochastic-combinatorial","title":"Sequential Stochastic Combinatorial Optimization Using Hierarchal Reinforcement Learning","date":"2025-02-08","arxiv_id":"2502.05537","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-regularized-diffusion-policy","title":"Behavior-Regularized Diffusion Policy Optimization for Offline Reinforcement Learning","date":"2025-02-07","arxiv_id":"2502.04778","repositories_listed":0,"syntology":null},{"url":null,"slug":"agency-is-frame-dependent","title":"Agency Is Frame-Dependent","date":"2025-02-06","arxiv_id":"2502.04403","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavioral-entropy-guided-dataset-generation","title":"Behavioral Entropy-Guided Dataset Generation for Offline Reinforcement Learning","date":"2025-02-06","arxiv_id":"2502.04141","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairness-aware-reinforcement-learning-via","title":"Fairness Aware Reinforcement Learning via Proximal Policy Optimization","date":"2025-02-06","arxiv_id":"2502.03953","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-on-aya-dyads-to","title":"Reinforcement Learning on Dyads to Enhance Medication Adherence","date":"2025-02-06","arxiv_id":"2502.06835","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-distillation-network-for-multi-agent","title":"Double Distillation Network for Multi-Agent Reinforcement Learning","date":"2025-02-05","arxiv_id":"2502.03125","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnirl-in-context-reinforcement-learning-by","title":"OmniRL: In-Context Reinforcement Learning by Large-Scale Meta-Training in Randomized Worlds","date":"2025-02-05","arxiv_id":"2502.02869","repositories_listed":0,"syntology":null},{"url":null,"slug":"teaching-language-models-to-critique-via","title":"Teaching Language Models to Critique via Reinforcement Learning","date":"2025-02-05","arxiv_id":"2502.03492","repositories_listed":0,"syntology":null},{"url":null,"slug":"ch-marl-constrained-hierarchical-multiagent","title":"CH-MARL: Constrained Hierarchical Multiagent Reinforcement Learning for Sustainable Maritime Logistics","date":"2025-02-04","arxiv_id":"2502.02060","repositories_listed":0,"syntology":null},{"url":null,"slug":"dhp-discrete-hierarchical-planning-for","title":"DHP: Discrete Hierarchical Planning for Hierarchical Reinforcement Learning Agents","date":"2025-02-04","arxiv_id":"2502.01956","repositories_listed":0,"syntology":null},{"url":null,"slug":"dime-diffusion-based-maximum-entropy","title":"DIME:Diffusion-Based Maximum Entropy Reinforcement Learning","date":"2025-02-04","arxiv_id":"2502.02316","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-guided-causal-state-representation-for","title":"Policy-Guided Causal State Representation for Offline Reinforcement Learning Recommendation","date":"2025-02-04","arxiv_id":"2502.02327","repositories_listed":0,"syntology":null},{"url":null,"slug":"acecoder-acing-coder-rl-via-automated-test","title":"ACECODER: Acing Coder RL via Automated Test-Case Synthesis","date":"2025-02-03","arxiv_id":"2502.01718","repositories_listed":0,"syntology":null},{"url":null,"slug":"competitive-programming-with-large-reasoning","title":"Competitive Programming with Large Reasoning Models","date":"2025-02-03","arxiv_id":"2502.06807","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploratory-utility-maximization-problem-with","title":"Exploratory Utility Maximization Problem with Tsallis Entropy","date":"2025-02-03","arxiv_id":"2502.01269","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-lanczos-method-for-systematic","title":"Generalized Lanczos method for systematic optimization of neural-network quantum states","date":"2025-02-03","arxiv_id":"2502.01264","repositories_listed":0,"syntology":null},{"url":null,"slug":"preference-vlm-leveraging-vlms-for-scalable","title":"Preference VLM: Leveraging VLMs for Scalable Preference-Based Reinforcement Learning","date":"2025-02-03","arxiv_id":"2502.01616","repositories_listed":0,"syntology":null},{"url":null,"slug":"process-supervised-reinforcement-learning-for","title":"Process-Supervised Reinforcement Learning for Code Generation","date":"2025-02-03","arxiv_id":"2502.01715","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-long-horizon","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","date":"2025-02-03","arxiv_id":"2502.01600","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-segment-feedback","title":"Reinforcement Learning with Segment Feedback","date":"2025-02-03","arxiv_id":"2502.01876","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-differences-between-direct-alignment","title":"The Differences Between Direct Alignment Algorithms are a Blur","date":"2025-02-03","arxiv_id":"2502.01237","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-task-generalization-via-memory","title":"Toward Task Generalization via Memory Augmentation in Meta-Reinforcement Learning","date":"2025-02-03","arxiv_id":"2502.01521","repositories_listed":0,"syntology":null},{"url":null,"slug":"vr-robo-a-real-to-sim-to-real-framework-for","title":"VR-Robo: A Real-to-Sim-to-Real Framework for Visual Robot Navigation and Locomotion","date":"2025-02-03","arxiv_id":"2502.01536","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-concept-based-neuron-level","title":"Compositional Concept-Based Neuron-Level Interpretability for Deep Reinforcement Learning","date":"2025-02-02","arxiv_id":"2502.00684","repositories_listed":0,"syntology":null},{"url":null,"slug":"decorrelated-soft-actor-critic-for-efficient","title":"Decorrelated Soft Actor-Critic for Efficient Deep Reinforcement Learning","date":"2025-01-31","arxiv_id":"2501.19133","repositories_listed":0,"syntology":null},{"url":null,"slug":"shaping-sparse-rewards-in-reinforcement","title":"Shaping Sparse Rewards in Reinforcement Learning: A Semi-supervised Approach","date":"2025-01-31","arxiv_id":"2501.19128","repositories_listed":0,"syntology":null},{"url":null,"slug":"spikingsoft-a-spiking-neuron-controller-for","title":"SpikingSoft: A Spiking Neuron Controller for Bio-inspired Locomotion with Soft Snake Robots","date":"2025-01-31","arxiv_id":"2501.19072","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-group-relative-policy-optimization-a","title":"Hybrid Group Relative Policy Optimization: A Multi-Sample Approach to Enhancing Policy Optimization","date":"2025-01-30","arxiv_id":"2502.01652","repositories_listed":0,"syntology":null},{"url":null,"slug":"certificated-actor-critic-hierarchical","title":"Certificated Actor-Critic: Hierarchical Reinforcement Learning with Control Barrier Functions for Safe Navigation","date":"2025-01-29","arxiv_id":"2501.17424","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-twin-enabled-real-time-control-in","title":"Digital Twin Synchronization: Bridging the Sim-RL Agent to a Real-Time Robotic Additive Manufacturing Control","date":"2025-01-29","arxiv_id":"2501.18016","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-portfolio-allocation","title":"Reinforcement-Learning Portfolio Allocation with Dynamic Embedding of Market Information","date":"2025-01-29","arxiv_id":"2501.17992","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-m-factor-a-novel-metric-for-evaluating","title":"The M-factor: A Novel Metric for Evaluating Neural Architecture Search in Resource-Constrained Environments","date":"2025-01-29","arxiv_id":"2501.17361","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-ensemble-models-based-on-graph","title":"Applying Ensemble Models based on Graph Neural Network and Reinforcement Learning for Wind Power Forecasting","date":"2025-01-28","arxiv_id":"2501.16591","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-vision-language-action-model-with","title":"Improving Vision-Language-Action Model with Online Reinforcement Learning","date":"2025-01-28","arxiv_id":"2501.16664","repositories_listed":0,"syntology":null},{"url":null,"slug":"induced-modularity-and-community-detection","title":"Induced Modularity and Community Detection for Functionally Interpretable Reinforcement Learning","date":"2025-01-28","arxiv_id":"2501.17077","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-interplay-between-sparsity-and","title":"On the Interplay Between Sparsity and Training in Deep Reinforcement Learning","date":"2025-01-28","arxiv_id":"2501.16729","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-real-world","title":"Safe Reinforcement Learning for Real-World Engine Control","date":"2025-01-28","arxiv_id":"2501.16613","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-ai-for-lyapunov-optimization","title":"Generative AI for Lyapunov Optimization Theory in UAV-based Low-Altitude Economy Networking","date":"2025-01-27","arxiv_id":"2501.15928","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-quantum-circuit","title":"Reinforcement Learning for Quantum Circuit Design: Using Matrix Representations","date":"2025-01-27","arxiv_id":"2501.16509","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-experience-sharing-in-reinforcement","title":"Selective Experience Sharing in Reinforcement Learning Enhances Interference Management","date":"2025-01-27","arxiv_id":"2501.15735","repositories_listed":0,"syntology":null},{"url":"/paper/towards-general-purpose-model-free","slug":"towards-general-purpose-model-free","title":"Towards General-Purpose Model-Free Reinforcement Learning","date":"2025-01-27","arxiv_id":"2501.16142","repositories_listed":0,"syntology":{"n":9,"n_ran":5,"n_constructed":4,"n_ran_checked":4,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/towards-general-purpose-model-free#ran","syntology_url":"https://syntology.ai/paper/2501.16142","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.16142"}},"official":null}},{"url":null,"slug":"advancing-tdfn-precise-fixation-point","title":"Advancing TDFN: Precise Fixation Point Generation Using Reconstruction Differences","date":"2025-01-26","arxiv_id":"2501.15603","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-knowledge-sharing-in-multi-agent","title":"Contextual Knowledge Sharing in Multi-Agent Reinforcement Learning with Decentralized Communication and Coordination","date":"2025-01-26","arxiv_id":"2501.15695","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-center-cooling-system-optimization-using","title":"Data Center Cooling System Optimization Using Offline Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15085","repositories_listed":0,"syntology":null},{"url":null,"slug":"extensive-exploration-in-complex-traffic","title":"Extensive Exploration in Complex Traffic Scenarios using Hierarchical Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.14992","repositories_listed":0,"syntology":null},{"url":null,"slug":"faster-machine-translation-ensembling-with","title":"Faster Machine Translation Ensembling with Reinforcement Learning and Competitive Correction","date":"2025-01-25","arxiv_id":"2501.15219","repositories_listed":0,"syntology":null},{"url":null,"slug":"music-generation-using-human-in-the-loop","title":"Music Generation using Human-In-The-Loop Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15304","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-lagrangian-optimization-for","title":"Predictive Lagrangian Optimization for Constrained Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15217","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-efficient-returns","title":"Reinforcement Learning for Efficient Returns Management","date":"2025-01-24","arxiv_id":"2501.14394","repositories_listed":0,"syntology":null}],"record_sha256":"6c42984d77babd7aaa837ac3e89d0481877387d8bc7f5f5c04a2dbdbb52fc864","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}