{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/49","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":49,"pages_in_order":135,"rows_per_page":100,"rows":[4801,4900],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/48","next":"/task/reinforcement-learning-2/papers/50","papers":[{"url":null,"slug":"plasticity-loss-in-deep-reinforcement","title":"Plasticity Loss in Deep Reinforcement Learning: A Survey","date":"2024-11-07","arxiv_id":"2411.04832","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-equivariance-in-reinforcement","title":"Approximate Equivariance in Reinforcement Learning","date":"2024-11-06","arxiv_id":"2411.04225","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-novice-to-expert-llm-agent-policy","title":"From Novice to Expert: LLM Agent Policy Optimization via Step-wise Reinforcement Learning","date":"2024-11-06","arxiv_id":"2411.03817","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-stationary-learning-of-neural-networks","title":"Non-Stationary Learning of Neural Networks with Automatic Soft Parameter Reset","date":"2024-11-06","arxiv_id":"2411.04034","repositories_listed":0,"syntology":null},{"url":null,"slug":"opportunities-of-reinforcement-learning-in","title":"Opportunities of Reinforcement Learning in South Africa's Just Transition","date":"2024-11-06","arxiv_id":"2411.15145","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-task-generalisation-with-multi","title":"Accelerating Task Generalisation with Multi-Level Skill Hierarchies","date":"2024-11-05","arxiv_id":"2411.02998","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-decision-making-for-uav","title":"Autonomous Decision Making for UAV Cooperative Pursuit-Evasion Game with Reinforcement Learning","date":"2024-11-05","arxiv_id":"2411.02983","repositories_listed":0,"syntology":null},{"url":"/paper/hierarchical-orchestra-of-policies","slug":"hierarchical-orchestra-of-policies","title":"Hierarchical Orchestra of Policies","date":"2024-11-05","arxiv_id":"2411.03008","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hierarchical-orchestra-of-policies#ran","syntology_url":"https://syntology.ai/paper/2411.03008","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03008"}},"official":null}},{"url":null,"slug":"when-to-localize-a-risk-constrained","title":"When to Localize? A Risk-Constrained Reinforcement Learning Approach","date":"2024-11-05","arxiv_id":"2411.02788","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-hidden-subgoals-under-temporal","title":"Learning Hidden Subgoals under Temporal Ordering Constraints in Reinforcement Learning","date":"2024-11-03","arxiv_id":"2411.01425","repositories_listed":0,"syntology":null},{"url":null,"slug":"teaching-models-to-improve-on-tape","title":"Teaching Models to Improve on Tape","date":"2024-11-03","arxiv_id":"2411.01483","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-reinforcement-learning-in","title":"A Review of Reinforcement Learning in Financial Applications","date":"2024-11-01","arxiv_id":"2411.12746","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-ml-model-versioning-in-edge","title":"Effective ML Model Versioning in Edge Networks","date":"2024-11-01","arxiv_id":"2411.01078","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-adaptive-mixed-criticality","title":"Enhancing Adaptive Mixed-Criticality Scheduling with Deep Reinforcement Learning","date":"2024-11-01","arxiv_id":"2411.00572","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-imitation-learning-based-optimal-energy","title":"Safe Imitation Learning-based Optimal Energy Storage Systems Dispatch in Distribution Networks","date":"2024-11-01","arxiv_id":"2411.00995","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistical-guarantees-for-lifelong","title":"Statistical Guarantees for Lifelong Reinforcement Learning using PAC-Bayes Theory","date":"2024-11-01","arxiv_id":"2411.00401","repositories_listed":0,"syntology":null},{"url":null,"slug":"anytime-constrained-multi-agent-reinforcement","title":"Anytime-Constrained Equilibria in Polynomial Time","date":"2024-10-31","arxiv_id":"2410.23637","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-automata-embeddings-for-goal","title":"Compositional Automata Embeddings for Goal-Conditioned Reinforcement Learning","date":"2024-10-31","arxiv_id":"2411.00205","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-entropy-hindsight-experience-replay","title":"Maximum Entropy Hindsight Experience Replay","date":"2024-10-31","arxiv_id":"2410.24016","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-safeguards-for-safe-and-model","title":"Progressive Safeguards for Safe and Model-Agnostic Reinforcement Learning","date":"2024-10-31","arxiv_id":"2410.24096","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-star-theoretical-analysis-of-reinforcement","title":"RL-STaR: Theoretical Analysis of Reinforcement Learning Frameworks for Self-Taught Reasoner","date":"2024-10-31","arxiv_id":"2410.23912","repositories_listed":0,"syntology":null},{"url":null,"slug":"grounding-by-trying-llms-with-reinforcement","title":"Grounding by Trying: LLMs with Reinforcement Learning-Enhanced Retrieval","date":"2024-10-30","arxiv_id":"2410.23214","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-governance-in-networked-systems-via","title":"Resource Governance in Networked Systems via Integrated Variational Autoencoders and Reinforcement Learning","date":"2024-10-30","arxiv_id":"2410.23393","repositories_listed":0,"syntology":null},{"url":null,"slug":"return-augmented-decision-transformer-for-off","title":"Return Augmented Decision Transformer for Off-Dynamics Reinforcement Learning","date":"2024-10-30","arxiv_id":"2410.23450","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-driving-car-racing-application-of-deep","title":"Self-Driving Car Racing: Application of Deep Reinforcement Learning","date":"2024-10-30","arxiv_id":"2410.22766","repositories_listed":0,"syntology":null},{"url":null,"slug":"stepping-out-of-the-shadows-reinforcement","title":"Stepping Out of the Shadows: Reinforcement Learning in Shadow Mode","date":"2024-10-30","arxiv_id":"2410.23419","repositories_listed":0,"syntology":null},{"url":null,"slug":"hindsight-experience-replay-accelerates","title":"Hindsight Experience Replay Accelerates Proximal Policy Optimization","date":"2024-10-29","arxiv_id":"2410.22524","repositories_listed":0,"syntology":null},{"url":null,"slug":"prefpaint-aligning-image-inpainting-diffusion","title":"PrefPaint: Aligning Image Inpainting Diffusion Model with Human Preference","date":"2024-10-29","arxiv_id":"2410.21966","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-minimum-cost-reach-avoid-using","title":"Solving Minimum-Cost Reach Avoid using Reinforcement Learning","date":"2024-10-29","arxiv_id":"2410.22600","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-reinforcement-learning-testbed","title":"A Multi-Agent Reinforcement Learning Testbed for Cognitive Radio Applications","date":"2024-10-28","arxiv_id":"2410.21521","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-legibility-in-multiagent-reinforcement","title":"Active Legibility in Multiagent Reinforcement Learning","date":"2024-10-28","arxiv_id":"2410.20954","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-constrained-policy-optimization","title":"Adversarial Constrained Policy Optimization: Improving Constrained Reinforcement Learning by Adapting Budgets","date":"2024-10-28","arxiv_id":"2410.20786","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-agent-deep-reinforcement-learning-for-1","title":"Dual-Agent Deep Reinforcement Learning for Dynamic Pricing and Replenishment","date":"2024-10-28","arxiv_id":"2410.21109","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-reinforcement-learning-for-incident","title":"Exploring reinforcement learning for incident response in autonomous military vehicles","date":"2024-10-28","arxiv_id":"2410.21407","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-9","title":"Offline Reinforcement Learning With Combinatorial Action Spaces","date":"2024-10-28","arxiv_id":"2410.21151","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-reinforcement-learning-based-two","title":"Quantum Reinforcement Learning-Based Two-Stage Unit Commitment Framework for Enhanced Power Systems Robustness","date":"2024-10-28","arxiv_id":"2410.21240","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-to-video-generative-adversarial-network","title":"Video to Video Generative Adversarial Network for Few-shot Learning Based on Policy Gradient","date":"2024-10-28","arxiv_id":"2410.20657","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-diversity-based-experience-replay","title":"Efficient Diversity-based Experience Replay for Deep Reinforcement Learning","date":"2024-10-27","arxiv_id":"2410.20487","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-the-sim-to-real-gap-leveraging","title":"Overcoming the Sim-to-Real Gap: Leveraging Simulation to Learn to Explore for Real-World RL","date":"2024-10-26","arxiv_id":"2410.20254","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-penalized-direct-preference","title":"Uncertainty-Penalized Direct Preference Optimization","date":"2024-10-26","arxiv_id":"2410.20187","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolving-choice-hysteresis-in-reinforcement","title":"Evolving choice hysteresis in reinforcement learning: comparing the adaptive value of positivity bias and gradual perseveration","date":"2024-10-25","arxiv_id":"2410.19434","repositories_listed":0,"syntology":null},{"url":null,"slug":"miles-making-imitation-learning-easy-with","title":"MILES: Making Imitation Learning Easy with Self-Supervision","date":"2024-10-25","arxiv_id":"2410.19693","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-with-9","title":"Multi-Agent Reinforcement Learning with Selective State-Space Models","date":"2024-10-25","arxiv_id":"2410.19382","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-to-online-multi-agent-reinforcement","title":"Offline-to-Online Multi-Agent Reinforcement Learning with Offline Value Function Memory and Sequential Exploration","date":"2024-10-25","arxiv_id":"2410.19450","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-robot-reinforcement-learning-with-goal","title":"On-Robot Reinforcement Learning with Goal-Contrastive Rewards","date":"2024-10-25","arxiv_id":"2410.19989","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-adaptive-average-reward","title":"Provably Adaptive Average Reward Reinforcement Learning for Metric Spaces","date":"2024-10-25","arxiv_id":"2410.19919","repositories_listed":0,"syntology":null},{"url":null,"slug":"sad-state-action-distillation-for-in-context","title":"Random Policy Enables In-Context Reinforcement Learning within Trust Horizons","date":"2024-10-25","arxiv_id":"2410.19982","repositories_listed":0,"syntology":null},{"url":null,"slug":"pointpatchrl-masked-reconstruction-improves","title":"PointPatchRL -- Masked Reconstruction Improves Reinforcement Learning on Point Clouds","date":"2024-10-24","arxiv_id":"2410.18800","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-the-chromatic","title":"Reinforcement Learning the Chromatic Symmetric Function","date":"2024-10-24","arxiv_id":"2410.19189","repositories_listed":0,"syntology":null},{"url":null,"slug":"samg-state-action-aware-offline-to-online","title":"SAMG: State-Action-Aware Offline-to-Online Reinforcement Learning with Offline Model Guidance","date":"2024-10-24","arxiv_id":"2410.18626","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-swarm-intelligence-and-reinforcement","title":"The Hive Mind is a Single Reinforcement Learning Agent","date":"2024-10-23","arxiv_id":"2410.17517","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-information-bottleneck-for-deep","title":"Multimodal Information Bottleneck for Deep Reinforcement Learning with Multiple Sensors","date":"2024-10-23","arxiv_id":"2410.17551","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-under-latent-dynamics","title":"Reinforcement Learning under Latent Dynamics: Toward Statistical and Algorithmic Modularity","date":"2024-10-23","arxiv_id":"2410.17904","repositories_listed":0,"syntology":null},{"url":null,"slug":"spire-synergistic-planning-imitation-and","title":"SPIRE: Synergistic Planning, Imitation, and Reinforcement Learning for Long-Horizon Manipulation","date":"2024-10-23","arxiv_id":"2410.18065","repositories_listed":0,"syntology":null},{"url":null,"slug":"drop-distributional-and-regular-optimism-and","title":"DROP: Distributional and Regular Optimism and Pessimism for Reinforcement Learning","date":"2024-10-22","arxiv_id":"2410.17473","repositories_listed":0,"syntology":null},{"url":null,"slug":"episodic-future-thinking-mechanism-for-multi","title":"Episodic Future Thinking Mechanism for Multi-agent Reinforcement Learning","date":"2024-10-22","arxiv_id":"2410.17373","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-in-context-preference-learning-using","title":"Large Language Models are In-context Preference Learners","date":"2024-10-22","arxiv_id":"2410.17233","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-multi-agent-reinforcement-2","title":"Hierarchical Multi-agent Reinforcement Learning for Cyber Network Defense","date":"2024-10-22","arxiv_id":"2410.17351","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-transformer-and-reinforcement","title":"Multi-Modal Transformer and Reinforcement Learning-based Beam Management","date":"2024-10-22","arxiv_id":"2410.19859","repositories_listed":0,"syntology":null},{"url":null,"slug":"quasinav-asymmetric-cost-aware-navigation","title":"QuasiNav: Asymmetric Cost-Aware Navigation Planning with Constrained Quasimetric Reinforcement Learning","date":"2024-10-22","arxiv_id":"2410.16666","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-curriculum-reinforcement","title":"Curriculum Reinforcement Learning for Complex Reward Functions","date":"2024-10-22","arxiv_id":"2410.16790","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-plug-and-play-fully-on-the-job-real-time","title":"A Plug-and-Play Fully On-the-Job Real-Time Reinforcement Learning Algorithm for a Direct-Drive Tandem-Wing Experiment Platforms Under Multiple Random Operating Conditions","date":"2024-10-21","arxiv_id":"2410.15554","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancements-in-electric-vehicle-charging","title":"Advancements in Electric Vehicle Charging Optimization: A Survey of Reinforcement Learning Approaches","date":"2024-10-21","arxiv_id":"2410.16425","repositories_listed":0,"syntology":null},{"url":null,"slug":"attentionpainter-an-efficient-and-adaptive","title":"AttentionPainter: An Efficient and Adaptive Stroke Predictor for Scene Painting","date":"2024-10-21","arxiv_id":"2410.16418","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-for-job-shop","title":"Offline reinforcement learning for job-shop scheduling problems","date":"2024-10-21","arxiv_id":"2410.15714","repositories_listed":0,"syntology":null},{"url":null,"slug":"rgmdt-return-gap-minimizing-decision-tree","title":"RGMDT: Return-Gap-Minimizing Decision Tree Extraction in Non-Euclidean Metric Space","date":"2024-10-21","arxiv_id":"2410.16517","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-and-alleviating-memory","title":"Understanding and Alleviating Memory Consumption in RLHF for LLMs","date":"2024-10-21","arxiv_id":"2410.15651","repositories_listed":0,"syntology":null},{"url":null,"slug":"assemblycomplete-3d-combinatorial","title":"AssemblyComplete: 3D Combinatorial Construction with Deep Reinforcement Learning","date":"2024-10-20","arxiv_id":"2410.15469","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-reinforcement-learning-model-for-post","title":"A Novel Reinforcement Learning Model for Post-Incident Malware Investigations","date":"2024-10-19","arxiv_id":"2410.15028","repositories_listed":0,"syntology":null},{"url":null,"slug":"gnnrl-smoothing-a-prior-free-reinforcement","title":"GNNRL-Smoothing: A Prior-Free Reinforcement Learning Model for Mesh Smoothing","date":"2024-10-19","arxiv_id":"2410.19834","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforced-trader-hrt-a-bi-level","title":"Hierarchical Reinforced Trader (HRT): A Bi-Level Approach for Optimizing Stock Selection and Execution","date":"2024-10-19","arxiv_id":"2410.14927","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-information-g-theory-for-range","title":"Semantic Information G Theory for Range Control with Tradeoff between Purposiveness and Efficiency","date":"2024-10-19","arxiv_id":"2411.05789","repositories_listed":0,"syntology":null},{"url":null,"slug":"harnessing-causality-in-reinforcement","title":"Harnessing Causality in Reinforcement Learning With Bagged Decision Times","date":"2024-10-18","arxiv_id":"2410.14659","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-end-to-end-neurosymbolic","title":"Interpretable end-to-end Neurosymbolic Reinforcement Learning agents","date":"2024-10-18","arxiv_id":"2410.14371","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-from-non","title":"Inverse Reinforcement Learning from Non-Stationary Learning Agents","date":"2024-10-18","arxiv_id":"2410.14135","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-reinforcement-learning-with-passive","title":"Online Reinforcement Learning with Passive Memory","date":"2024-10-18","arxiv_id":"2410.14665","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-non-markov-market","title":"Reinforcement Learning in Non-Markov Market-Making","date":"2024-10-18","arxiv_id":"2410.14504","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-reinforcement-learning-in","title":"Transfer Reinforcement Learning in Heterogeneous Action Spaces using Subgoal Mapping","date":"2024-10-18","arxiv_id":"2410.14484","repositories_listed":0,"syntology":null},{"url":null,"slug":"utilizing-large-language-models-for-event","title":"Utilizing Large Language Models for Event Deconstruction to Enhance Multimodal Aspect-Based Sentiment Analysis","date":"2024-10-18","arxiv_id":"2410.14150","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-inception-for-bounded-backdoor","title":"Adversarial Inception Backdoor Attacks against Reinforcement Learning","date":"2024-10-17","arxiv_id":"2410.13995","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximating-auction-equilibria-with","title":"Approximating Auction Equilibria with Reinforcement Learning","date":"2024-10-17","arxiv_id":"2410.13960","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-reinforcement-learning-for-robust","title":"Guided Reinforcement Learning for Robust Multi-Contact Loco-Manipulation","date":"2024-10-17","arxiv_id":"2410.13817","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-prior-free-black-box-non-stationary","title":"Is Prior-Free Black-Box Non-Stationary Reinforcement Learning Feasible?","date":"2024-10-17","arxiv_id":"2410.13772","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-optimal-transport-in-offline","title":"Rethinking Optimal Transport in Offline Reinforcement Learning","date":"2024-10-17","arxiv_id":"2410.14069","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-learning-rate-for-deep-reinforcement","title":"Dynamic Learning Rate for Deep Reinforcement Learning: A Bandit Approach","date":"2024-10-16","arxiv_id":"2410.12598","repositories_listed":0,"syntology":null},{"url":null,"slug":"gan-based-top-down-view-synthesis-in","title":"GAN Based Top-Down View Synthesis in Reinforcement Learning Environments","date":"2024-10-16","arxiv_id":"2410.12372","repositories_listed":0,"syntology":null},{"url":null,"slug":"insights-from-the-inverse-reconstructing-llm","title":"Insights from the Inverse: Reconstructing LLM Training Goals Through Inverse RL","date":"2024-10-16","arxiv_id":"2410.12491","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectrum-sharing-using-deep-reinforcement","title":"Spectrum Sharing using Deep Reinforcement Learning in Vehicular Networks","date":"2024-10-16","arxiv_id":"2410.12521","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-persistent-threats-apt-attribution","title":"Advanced Persistent Threats (APT) Attribution Using Deep Reinforcement Learning","date":"2024-10-15","arxiv_id":"2410.11463","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-unsupervised-skill-discovery-for","title":"Disentangled Unsupervised Skill Discovery for Efficient Hierarchical Reinforcement Learning","date":"2024-10-15","arxiv_id":"2410.11251","repositories_listed":0,"syntology":null},{"url":null,"slug":"dodt-enhanced-online-decision-transformer","title":"DODT: Enhanced Online Decision Transformer Learning through Dreamer's Actor-Critic Trajectory Forecasting","date":"2024-10-15","arxiv_id":"2410.11359","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-reinforcement-learning-a-tool","title":"Multi-objective Reinforcement Learning: A Tool for Pluralistic Alignment","date":"2024-10-15","arxiv_id":"2410.11221","repositories_listed":0,"syntology":null},{"url":null,"slug":"physical-informed-inspired-deep-reinforcement","title":"Physical Informed-Inspired Deep Reinforcement Learning Based Bi-Level Programming for Microgrid Scheduling","date":"2024-10-15","arxiv_id":"2410.11932","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-the-dynamic-volatility-fitting","title":"Solving The Dynamic Volatility Fitting Problem: A Deep Reinforcement Learning Approach","date":"2024-10-15","arxiv_id":"2410.11789","repositories_listed":0,"syntology":null},{"url":"/paper/action-gaps-and-advantages-in-continuous-time","slug":"action-gaps-and-advantages-in-continuous-time","title":"Action Gaps and Advantages in Continuous-Time Distributional Reinforcement Learning","date":"2024-10-14","arxiv_id":"2410.11022","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/action-gaps-and-advantages-in-continuous-time#ran","syntology_url":"https://syntology.ai/paper/2410.11022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11022"}},"official":null}},{"url":null,"slug":"burning-red-unlocking-subtask-driven","title":"Burning RED: Unlocking Subtask-Driven Reinforcement Learning and Risk-Awareness in Average-Reward Markov Decision Processes","date":"2024-10-14","arxiv_id":"2410.10578","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-shielding-and-reinforcement","title":"Compositional Shielding and Reinforcement Learning for Multi-Agent Systems","date":"2024-10-14","arxiv_id":"2410.10460","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversity-aware-reinforcement-learning-for-de","title":"Diversity-Aware Reinforcement Learning for de novo Drug Design","date":"2024-10-14","arxiv_id":"2410.10431","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-robustness-in-deep-reinforcement","title":"Enhancing Robustness in Deep Reinforcement Learning: A Lyapunov Exponent Approach","date":"2024-10-14","arxiv_id":"2410.10674","repositories_listed":0,"syntology":null},{"url":null,"slug":"qe-ebm-using-quality-estimators-as-energy","title":"QE-EBM: Using Quality Estimators as Energy Loss for Machine Translation","date":"2024-10-14","arxiv_id":"2410.10228","repositories_listed":0,"syntology":null}],"record_sha256":"5ede9e94a984bea5ec366965cdc2346e21a72083b9c4e131026595bdbd22e942","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}