{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/50","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":50,"pages_in_order":135,"rows_per_page":100,"rows":[4901,5000],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/49","next":"/task/reinforcement-learning-2/papers/51","papers":[{"url":null,"slug":"meta-reinforcement-learning-with-universal","title":"Meta-Reinforcement Learning with Universal Policy Adaptation: Provable Near-Optimality under All-task Optimum Comparator","date":"2024-10-13","arxiv_id":"2410.09728","repositories_listed":0,"syntology":null},{"url":null,"slug":"actsafe-active-exploration-with-safety","title":"ActSafe: Active Exploration with Safety Constraints for Reinforcement Learning","date":"2024-10-12","arxiv_id":"2410.09486","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-hyperbolic-spaces","title":"Reinforcement Learning in Hyperbolic Spaces: Models and Experiments","date":"2024-10-12","arxiv_id":"2410.09466","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-domain-specific-modelling","title":"Towards a Domain-Specific Modelling Environment for Reinforcement Learning","date":"2024-10-12","arxiv_id":"2410.09368","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-universal-value-function","title":"Hierarchical Universal Value Function Approximators","date":"2024-10-11","arxiv_id":"2410.08997","repositories_listed":0,"syntology":null},{"url":null,"slug":"sold-reinforcement-learning-with-slot-object","title":"SOLD: Slot Object-Centric Latent Dynamics Models for Relational Manipulation Learning from Pixels","date":"2024-10-11","arxiv_id":"2410.08822","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-with-large","title":"Efficient Reinforcement Learning with Large Language Model Priors","date":"2024-10-10","arxiv_id":"2410.07927","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuroplastic-expansion-in-deep-reinforcement","title":"Neuroplastic Expansion in Deep Reinforcement Learning","date":"2024-10-10","arxiv_id":"2410.07994","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-hierarchical-reinforcement-learning","title":"Offline Hierarchical Reinforcement Learning via Inverse Optimization","date":"2024-10-10","arxiv_id":"2410.07933","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-inverse-constrained-reinforcement","title":"Offline Inverse Constrained Reinforcement Learning for Safe-Critical Decision Making in Healthcare","date":"2024-10-10","arxiv_id":"2410.07525","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-grid-sampling-limit-sde","title":"On the grid-sampling limit SDE","date":"2024-10-10","arxiv_id":"2410.07778","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-satisfaction-of-temporal-logic","title":"Probabilistic Satisfaction of Temporal Logic Constraints in Reinforcement Learning via Adaptive Policy-Switching","date":"2024-10-10","arxiv_id":"2410.08022","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-adversarial-inverse-reinforcement-1","title":"On Reward Transferability in Adversarial Inverse Reinforcement Learning: Insights from Random Matrix Theory","date":"2024-10-10","arxiv_id":"2410.07643","repositories_listed":0,"syntology":null},{"url":null,"slug":"fostering-intrinsic-motivation-in","title":"Fostering Intrinsic Motivation in Reinforcement Learning with Pretrained Foundation Models","date":"2024-10-09","arxiv_id":"2410.07404","repositories_listed":0,"syntology":null},{"url":null,"slug":"honesty-to-subterfuge-in-context","title":"Honesty to Subterfuge: In-Context Reinforcement Learning Can Make Honest Models Reward Hack","date":"2024-10-09","arxiv_id":"2410.06491","repositories_listed":0,"syntology":null},{"url":null,"slug":"motionrl-align-text-to-motion-generation-to","title":"MotionRL: Align Text-to-Motion Generation to Human Preferences with Multi-Reward Reinforcement Learning","date":"2024-10-09","arxiv_id":"2410.06513","repositories_listed":0,"syntology":null},{"url":null,"slug":"reindiffuse-crafting-physically-plausible","title":"ReinDiffuse: Crafting Physically Plausible Motions with Reinforced Diffusion Model","date":"2024-10-09","arxiv_id":"2410.07296","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-for-a-class-of-cascade","title":"Transfer Learning for a Class of Cascade Dynamical Systems","date":"2024-10-09","arxiv_id":"2410.06828","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-policy-evaluation-with-safety","title":"Efficient Policy Evaluation with Safety Constraint for Reinforcement Learning","date":"2024-10-08","arxiv_id":"2410.05655","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-from-imperfect-1","title":"Reinforcement Learning From Imperfect Corrective Actions And Proxy Rewards","date":"2024-10-08","arxiv_id":"2410.05782","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlrf4rec-reinforcement-learning-from-recsys","title":"Direct Preference Optimization for LLM-Enhanced Recommendation Systems","date":"2024-10-08","arxiv_id":"2410.05939","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-multi-goal-robotic-tasks-with","title":"Solving Multi-Goal Robotic Tasks with Decision Transformer","date":"2024-10-08","arxiv_id":"2410.06347","repositories_listed":0,"syntology":null},{"url":null,"slug":"alpharouter-quantum-circuit-routing-with","title":"AlphaRouter: Quantum Circuit Routing with Reinforcement Learning and Tree Search","date":"2024-10-07","arxiv_id":"2410.05115","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-a-classifier-for-active-fire","title":"Designing a Classifier for Active Fire Detection from Multispectral Satellite Imagery Using Neural Architecture Search","date":"2024-10-07","arxiv_id":"2410.05425","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-model-based-reinforcement-learning-2","title":"Efficient Model-Based Reinforcement Learning Through Optimistic Thompson Sampling","date":"2024-10-07","arxiv_id":"2410.04988","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-feedback-efficient-reinforcement","title":"HERO: Human-Feedback Efficient Reinforcement Learning for Online Diffusion Model Finetuning","date":"2024-10-07","arxiv_id":"2410.05116","repositories_listed":0,"syntology":null},{"url":null,"slug":"mastering-chinese-chess-ai-xiangqi-without","title":"Mastering Chinese Chess AI (Xiangqi) Without Search","date":"2024-10-07","arxiv_id":"2410.04865","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-using-reinforcement-learning-for","title":"Towards using Reinforcement Learning for Scaling and Data Replication in Cloud Systems","date":"2024-10-07","arxiv_id":"2410.11862","repositories_listed":0,"syntology":null},{"url":null,"slug":"adamemento-adaptive-memory-assisted-policy","title":"AdaMemento: Adaptive Memory-Assisted Policy Optimization for Reinforcement Learning","date":"2024-10-06","arxiv_id":"2410.04498","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-under-frequency-load-shedding","title":"Data-driven Under Frequency Load Shedding Using Reinforcement Learning","date":"2024-10-06","arxiv_id":"2410.04316","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-debugging-deep-reinforcement-learning","title":"Toward Debugging Deep Reinforcement Learning Programs with RLExplorer","date":"2024-10-06","arxiv_id":"2410.04322","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-action-priors-from-a-single-gait-cycle","title":"Latent Action Priors for Locomotion with Deep Reinforcement Learning","date":"2024-10-04","arxiv_id":"2410.03246","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-on-more-reachable-tasks-for","title":"Training on more Reachable Tasks for Generalisation in Reinforcement Learning","date":"2024-10-04","arxiv_id":"2410.03565","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-embodiment-dexterous-grasping-with","title":"Cross-Embodiment Dexterous Grasping with Reinforcement Learning","date":"2024-10-03","arxiv_id":"2410.02479","repositories_listed":0,"syntology":null},{"url":null,"slug":"doubly-optimal-policy-evaluation-for","title":"Doubly Optimal Policy Evaluation for Reinforcement Learning","date":"2024-10-03","arxiv_id":"2410.02226","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-active-learning-for-reinforcement","title":"Dual Active Learning for Reinforcement Learning from Human Feedback","date":"2024-10-03","arxiv_id":"2410.02504","repositories_listed":0,"syntology":null},{"url":null,"slug":"comadice-offline-cooperative-multi-agent","title":"ComaDICE: Offline Cooperative Multi-Agent Reinforcement Learning with Stationary Distribution Shift Regularization","date":"2024-10-02","arxiv_id":"2410.01954","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-path-and-cycle-counting-formulae-in","title":"Finding path and cycle counting formulae in graphs with Deep Reinforcement Learning","date":"2024-10-02","arxiv_id":"2410.01661","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-reward-models","title":"Generative Reward Models","date":"2024-10-02","arxiv_id":"2410.12832","repositories_listed":0,"syntology":null},{"url":null,"slug":"hidden-in-plain-text-emergence-mitigation-of","title":"Hidden in Plain Text: Emergence & Mitigation of Steganographic Collusion in LLMs","date":"2024-10-02","arxiv_id":"2410.03768","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-augmented-symbolic-reinforcement-learning","title":"LLM-Augmented Symbolic Reinforcement Learning with Landmark-Based Task Decomposition","date":"2024-10-02","arxiv_id":"2410.01929","repositories_listed":0,"syntology":null},{"url":null,"slug":"performant-memory-efficient-and-scalable","title":"Sable: a Performant, Efficient and Scalable Sequence Model for MARL","date":"2024-10-02","arxiv_id":"2410.01706","repositories_listed":0,"syntology":null},{"url":null,"slug":"prend-enhancing-intrinsic-motivation-in","title":"PreND: Enhancing Intrinsic Motivation in Reinforcement Learning through Pre-trained Network Distillation","date":"2024-10-02","arxiv_id":"2410.01745","repositories_listed":0,"syntology":null},{"url":null,"slug":"realizable-continuous-space-shields-for-safe","title":"Realizable Continuous-Space Shields for Safe Reinforcement Learning","date":"2024-10-02","arxiv_id":"2410.02038","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlef-grounding-code-llms-in-execution","title":"RLEF: Grounding Code LLMs in Execution Feedback with Reinforcement Learning","date":"2024-10-02","arxiv_id":"2410.02089","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-reinforcement-learning-based-neural-1","title":"Scalable Reinforcement Learning-based Neural Architecture Search","date":"2024-10-02","arxiv_id":"2410.01431","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-abstraction-for-reinforcement","title":"Contrastive Abstraction for Reinforcement Learning","date":"2024-10-01","arxiv_id":"2410.00704","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficient-computation-with-dvfs-using","title":"Energy-Efficient Computation with DVFS using Deep Reinforcement Learning for Multi-Task Systems in Edge Computing","date":"2024-09-28","arxiv_id":"2409.19434","repositories_listed":0,"syntology":null},{"url":null,"slug":"refutation-of-spectral-graph-theory-1","title":"Refutation of Spectral Graph Theory Conjectures with Search Algorithms)","date":"2024-09-27","arxiv_id":"2409.18626","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-deep-reinforcement-learning-for-volt","title":"Robust Deep Reinforcement Learning for Volt-VAR Optimization in Active Distribution System under Uncertainty","date":"2024-09-27","arxiv_id":"2409.18937","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-free-reinforcement-learning","title":"State-free Reinforcement Learning","date":"2024-09-27","arxiv_id":"2409.18439","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-neural-architecture-search-based","title":"A Survey on Neural Architecture Search Based on Reinforcement Learning","date":"2024-09-26","arxiv_id":"2409.18163","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-multi-trait-essay-scoring-via","title":"Autoregressive Multi-trait Essay Scoring via Reinforcement Learning with Scoring-aware Multiple Rewards","date":"2024-09-26","arxiv_id":"2409.17472","repositories_listed":0,"syntology":null},{"url":null,"slug":"criticality-and-safety-margins-for","title":"Criticality and Safety Margins for Reinforcement Learning","date":"2024-09-26","arxiv_id":"2409.18289","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-with-multiple-1","title":"Inverse Reinforcement Learning with Multiple Planning Horizons","date":"2024-09-26","arxiv_id":"2409.18051","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-versus-model-based-reinforcement","title":"Model-Free versus Model-Based Reinforcement Learning for Fixed-Wing UAV Attitude Control Under Varying Wind Conditions","date":"2024-09-26","arxiv_id":"2409.17896","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigation-in-a-simplified-urban-flow-through","title":"Navigation in a simplified Urban Flow through Deep Reinforcement Learning","date":"2024-09-26","arxiv_id":"2409.17922","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-random-measure-approach-to-reinforcement","title":"A random measure approach to reinforcement learning in continuous time","date":"2024-09-25","arxiv_id":"2409.17200","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-for-deep-reinforcement-learning","title":"A Survey for Deep Reinforcement Learning Based Network Intrusion Detection","date":"2024-09-25","arxiv_id":"2410.07612","repositories_listed":0,"syntology":null},{"url":"/paper/exploring-semantic-clustering-in-deep","slug":"exploring-semantic-clustering-in-deep","title":"Exploring Semantic Clustering in Deep Reinforcement Learning for Video Games","date":"2024-09-25","arxiv_id":"2409.17411","repositories_listed":0,"syntology":{"n":17,"n_ran":9,"n_constructed":6,"n_ran_checked":7,"n_instrument":2,"n_unverified":8,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/exploring-semantic-clustering-in-deep#ran","syntology_url":"https://syntology.ai/paper/2409.17411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.17411"}},"official":null}},{"url":null,"slug":"learning-bipedal-walking-for-humanoid-robots","title":"Learning Bipedal Walking for Humanoid Robots in Challenging Environments with Obstacle Avoidance","date":"2024-09-25","arxiv_id":"2410.08212","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-and-distributional-reinforcement","title":"Offline and Distributional Reinforcement Learning for Radio Resource Management","date":"2024-09-25","arxiv_id":"2409.16764","repositories_listed":0,"syntology":null},{"url":null,"slug":"offripp-offline-rl-based-informative-path","title":"OffRIPP: Offline RL-based Informative Path Planning","date":"2024-09-25","arxiv_id":"2409.16830","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-finite-space-mean","title":"Reinforcement Learning for Finite Space Mean-Field Type Games","date":"2024-09-25","arxiv_id":"2409.18152","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-space-mission-planning-a","title":"Revisiting Space Mission Planning: A Reinforcement Learning-Guided Approach for Multi-Debris Rendezvous","date":"2024-09-25","arxiv_id":"2409.16882","repositories_listed":0,"syntology":null},{"url":null,"slug":"symbolic-state-partition-for-reinforcement","title":"Symbolic State Partitioning for Reinforcement Learning","date":"2024-09-25","arxiv_id":"2409.16791","repositories_listed":0,"syntology":null},{"url":null,"slug":"topological-foundations-of-reinforcement","title":"Topological Foundations of Reinforcement Learning","date":"2024-09-25","arxiv_id":"2410.03706","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-representations-in-state-space","title":"Uncertainty Representations in State-Space Layers for Deep Reinforcement Learning under Partial Observability","date":"2024-09-25","arxiv_id":"2409.16824","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-critical-review-of-safe-reinforcement","title":"A Critical Review of Safe Reinforcement Learning Techniques in Smart Grid Applications","date":"2024-09-24","arxiv_id":"2409.16256","repositories_listed":0,"syntology":null},{"url":null,"slug":"clsp-high-fidelity-contrastive-language-state","title":"CLSP: High-Fidelity Contrastive Language-State Pre-training for Agent State Representation","date":"2024-09-24","arxiv_id":"2409.15806","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-reward-model-noise-in-instruction","title":"The Dark Side of Rich Rewards: Understanding and Mitigating Noise in VLM Rewards","date":"2024-09-24","arxiv_id":"2409.15922","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-exploration-in-inverse","title":"Provably Efficient Exploration in Inverse Constrained Reinforcement Learning","date":"2024-09-24","arxiv_id":"2409.15963","repositories_listed":0,"syntology":null},{"url":null,"slug":"surgirl-towards-life-long-learning-for","title":"SurgIRL: Towards Life-Long Learning for Surgical Automation by Incremental Reinforcement Learning","date":"2024-09-24","arxiv_id":"2409.15651","repositories_listed":0,"syntology":null},{"url":null,"slug":"acting-for-the-right-reasons-creating-reason","title":"Acting for the Right Reasons: Creating Reason-Sensitive Artificial Moral Agents","date":"2024-09-23","arxiv_id":"2409.15014","repositories_listed":0,"syntology":null},{"url":null,"slug":"candere-coach-reinforcement-learning-from","title":"CANDERE-COACH: Reinforcement Learning from Noisy Feedback","date":"2024-09-23","arxiv_id":"2409.15521","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-obstacle","title":"Deep Reinforcement Learning-based Obstacle Avoidance for Robot Movement in Warehouse Environments","date":"2024-09-23","arxiv_id":"2409.14972","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-routing-algorithm-over-sdn","title":"Intelligent Routing Algorithm over SDN: Reusable Reinforcement Learning Approach","date":"2024-09-23","arxiv_id":"2409.15226","repositories_listed":0,"syntology":null},{"url":null,"slug":"cosbo-conservative-offline-simulation-based","title":"COSBO: Conservative Offline Simulation-Based Policy Optimization","date":"2024-09-22","arxiv_id":"2409.14412","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-multi-agent-reinforcement-learning-5","title":"Scalable Multi-agent Reinforcement Learning for Factory-wide Dynamic Scheduling","date":"2024-09-20","arxiv_id":"2409.13571","repositories_listed":0,"syntology":null},{"url":null,"slug":"soloparkour-constrained-reinforcement","title":"SoloParkour: Constrained Reinforcement Learning for Visual Locomotion from Privileged Experience","date":"2024-09-20","arxiv_id":"2409.13678","repositories_listed":0,"syntology":null},{"url":null,"slug":"taco-rl-task-aware-prompt-compression","title":"TACO-RL: Task Aware Prompt Compression Optimization with Reinforcement Learning","date":"2024-09-19","arxiv_id":"2409.13035","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-central-role-of-the-loss-function-in","title":"The Central Role of the Loss Function in Reinforcement Learning","date":"2024-09-19","arxiv_id":"2409.12799","repositories_listed":0,"syntology":null},{"url":null,"slug":"harp-human-assisted-regrouping-with","title":"HARP: Human-Assisted Regrouping with Permutation Invariant Critic for Multi-Agent Reinforcement Learning","date":"2024-09-18","arxiv_id":"2409.11741","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-lie-group","title":"Reinforcement Learning with Lie Group Orientations for Robotics","date":"2024-09-18","arxiv_id":"2409.11935","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-environment-for-4","title":"A Reinforcement Learning Environment for Automatic Code Optimization in the MLIR Compiler","date":"2024-09-17","arxiv_id":"2409.11068","repositories_listed":0,"syntology":null},{"url":null,"slug":"attacking-slicing-network-via-side-channel","title":"Attacking Slicing Network via Side-channel Reinforcement Learning Attack","date":"2024-09-17","arxiv_id":"2409.11258","repositories_listed":0,"syntology":null},{"url":null,"slug":"linear-jamming-bandits-learning-to-jam-5g","title":"Linear Jamming Bandits: Learning to Jam 5G-based Coded Communications Systems","date":"2024-09-17","arxiv_id":"2409.11191","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-policy-actor-critic-reinforcement-learning","title":"On-policy Actor-Critic Reinforcement Learning for Multi-UAV Exploration","date":"2024-09-17","arxiv_id":"2409.11058","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-model-free-optimal-control-method-with","title":"A Model-Free Optimal Control Method With Fixed Terminal States and Delay","date":"2024-09-16","arxiv_id":"2409.10722","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-uncertainty-for-safe-social","title":"Disentangling Uncertainty for Safe Social Navigation using Deep Reinforcement Learning","date":"2024-09-16","arxiv_id":"2409.10655","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-statistical","title":"Reinforcement learning-based statistical search strategy for an axion model from flavor","date":"2024-09-16","arxiv_id":"2409.10023","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-quasi-hyperbolic","title":"Reinforcement Learning with Quasi-Hyperbolic Discounting","date":"2024-09-16","arxiv_id":"2409.10583","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-oriented-pruning-and-interpretation-of","title":"Safety-Oriented Pruning and Interpretation of Reinforcement Learning Policies","date":"2024-09-16","arxiv_id":"2409.10218","repositories_listed":0,"syntology":null},{"url":null,"slug":"shire-enhancing-sample-efficiency-using-human","title":"SHIRE: Enhancing Sample Efficiency using Human Intuition in REinforcement Learning","date":"2024-09-16","arxiv_id":"2409.09990","repositories_listed":0,"syntology":null},{"url":null,"slug":"critic-as-lyapunov-function-calf-a-model-free","title":"Critic as Lyapunov function (CALF): a model-free, stability-ensuring agent","date":"2024-09-15","arxiv_id":"2409.09869","repositories_listed":0,"syntology":null},{"url":null,"slug":"kan-v-s-mlp-for-offline-reinforcement","title":"KAN v.s. MLP for Offline Reinforcement Learning","date":"2024-09-15","arxiv_id":"2409.09653","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-dimensionality-in-2d-rectangle","title":"Mitigating Dimensionality in 2D Rectangle Packing Problem under Reinforcement Learning Schema","date":"2024-09-15","arxiv_id":"2409.09677","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-transformer-long-horizon-offline","title":"Planning Transformer: Long-Horizon Offline Reinforcement Learning with Planning Tokens","date":"2024-09-14","arxiv_id":"2409.09513","repositories_listed":0,"syntology":null},{"url":null,"slug":"curricula-for-learning-robust-policies-over","title":"Curricula for Learning Robust Policies over Factored State Representations in Changing Environments","date":"2024-09-13","arxiv_id":"2409.09169","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-inspired-reinforcement-learning-for","title":"Quantum-inspired Reinforcement Learning for Synthesizable Drug Design","date":"2024-09-13","arxiv_id":"2409.09183","repositories_listed":0,"syntology":null}],"record_sha256":"f9ac066281abb01e4ae303bd808db4485081fc14a0c4b7b07487e637064f26f6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}