{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/79","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":79,"pages_in_order":152,"rows_per_page":100,"rows":[7801,7900],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/78","next":"/task/reinforcement-learning-1/papers/80","papers":[{"url":null,"slug":"metaems-a-meta-reinforcement-learning-based","title":"MetaEMS: A Meta Reinforcement Learning-based Control Framework for Building Energy Management System","date":"2022-10-23","arxiv_id":"2210.12590","repositories_listed":0,"syntology":null},{"url":null,"slug":"attitude-control-of-highly-maneuverable","title":"Attitude Control of Highly Maneuverable Aircraft Using an Improved Q-learning","date":"2022-10-22","arxiv_id":"2210.12317","repositories_listed":0,"syntology":null},{"url":null,"slug":"faster-and-more-diverse-de-novo-molecular","title":"Faster and more diverse de novo molecular optimization with double-loop reinforcement learning using augmented SMILES","date":"2022-10-22","arxiv_id":"2210.12458","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-transfer-in-deep-reinforcement","title":"Probing Transfer in Deep Reinforcement Learning without Task Engineering","date":"2022-10-22","arxiv_id":"2210.12448","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-reinforcement-learning-with-group","title":"Continual Vision-based Reinforcement Learning with Group Symmetries","date":"2022-10-21","arxiv_id":"2210.12301","repositories_listed":0,"syntology":null},{"url":null,"slug":"counterfactual-explanations-for-reinforcement","title":"Redefining Counterfactual Explanations for Reinforcement Learning: Overview, Challenges and Opportunities","date":"2022-10-21","arxiv_id":"2210.11846","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-inverse","title":"Deep Reinforcement Learning for Inverse Inorganic Materials Design","date":"2022-10-21","arxiv_id":"2210.11931","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-stabilization","title":"Deep Reinforcement Learning for Stabilization of Large-scale Probabilistic Boolean Networks","date":"2022-10-21","arxiv_id":"2210.12229","repositories_listed":0,"syntology":null},{"url":null,"slug":"group-distributionally-robust-reinforcement","title":"Group Distributionally Robust Reinforcement Learning with Hierarchical Latent Variables","date":"2022-10-21","arxiv_id":"2210.12262","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-offline-reinforcement-learning-via","title":"Implicit Offline Reinforcement Learning via Supervised Learning","date":"2022-10-21","arxiv_id":"2210.12272","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-policy-summaries-with-reward","title":"Integrating Policy Summaries with Reward Decomposition for Explaining Reinforcement Learning Agents","date":"2022-10-21","arxiv_id":"2210.11825","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-connection-between-bregman-divergence","title":"On the connection between Bregman divergence and value in regularized Markov decision processes","date":"2022-10-21","arxiv_id":"2210.12160","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-with-uncertainty-deep-exploration-in","title":"Epistemic Monte Carlo Tree Search","date":"2022-10-21","arxiv_id":"2210.13455","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-quantum-enabled-6g-slicing","title":"Towards Quantum-Enabled 6G Slicing","date":"2022-10-21","arxiv_id":"2212.11755","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-session-recommendations-in-e","title":"Fine-Grained Session Recommendations in E-commerce using Deep Reinforcement Learning","date":"2022-10-20","arxiv_id":"2210.15451","repositories_listed":0,"syntology":null},{"url":null,"slug":"horizon-free-reinforcement-learning-for","title":"Horizon-Free and Variance-Dependent Reinforcement Learning for Latent Markov Decision Processes","date":"2022-10-20","arxiv_id":"2210.11604","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-imitation-via-mirror-descent-inverse-1","title":"Robust Imitation via Mirror Descent Inverse Reinforcement Learning","date":"2022-10-20","arxiv_id":"2210.11201","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-policy-improvement-in-constrained-markov","title":"Safe Policy Improvement in Constrained Markov Decision Processes","date":"2022-10-20","arxiv_id":"2210.11259","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-in-multi","title":"A Reinforcement Learning Approach in Multi-Phase Second-Price Auction Design","date":"2022-10-19","arxiv_id":"2210.10278","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-7","title":"Hierarchical Reinforcement Learning for Furniture Layout in Virtual Indoor Scenes","date":"2022-10-19","arxiv_id":"2210.10431","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrated-decision-and-control-for-high","title":"Integrated Decision and Control for High-Level Automated Vehicles by Mixed Policy Gradient and Its Experiment Verification","date":"2022-10-19","arxiv_id":"2210.10613","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-power-of-pre-training-for","title":"On the Power of Pre-training for Generalization in RL: Provable Benefits and Hardness","date":"2022-10-19","arxiv_id":"2210.10464","repositories_listed":0,"syntology":null},{"url":null,"slug":"oracles-followers-stackelberg-equilibria-in","title":"Oracles & Followers: Stackelberg Equilibria in Deep Multi-Agent Reinforcement Learning","date":"2022-10-19","arxiv_id":"2210.11942","repositories_listed":0,"syntology":null},{"url":null,"slug":"palm-up-playing-in-the-latent-manifold-for","title":"Palm up: Playing in the Latent Manifold for Unsupervised Pretraining","date":"2022-10-19","arxiv_id":"2210.10913","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-safe-reinforcement-learning-via","title":"Provably Safe Reinforcement Learning via Action Projection using Reachability Analysis and Polynomial Zonotopes","date":"2022-10-19","arxiv_id":"2210.10691","repositories_listed":0,"syntology":null},{"url":null,"slug":"robot-navigation-with-reinforcement-learned","title":"Robot Navigation with Reinforcement Learned Path Generation and Fine-Tuned Motion Control","date":"2022-10-19","arxiv_id":"2210.10639","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-table-wiping-via-reinforcement","title":"Robotic Table Wiping via Reinforcement Learning and Whole-body Trajectory Optimization","date":"2022-10-19","arxiv_id":"2210.10865","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-laws-for-reward-model","title":"Scaling Laws for Reward Model Overoptimization","date":"2022-10-19","arxiv_id":"2210.10760","repositories_listed":0,"syntology":null},{"url":null,"slug":"rpm-generalizable-behaviors-for-multi-agent","title":"RPM: Generalizable Behaviors for Multi-Agent Reinforcement Learning","date":"2022-10-18","arxiv_id":"2210.09646","repositories_listed":0,"syntology":null},{"url":null,"slug":"unpacking-reward-shaping-understanding-the","title":"Unpacking Reward Shaping: Understanding the Benefits of Reward Engineering on Sample Complexity","date":"2022-10-18","arxiv_id":"2210.09579","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-offline-reinforcement-learning-via","title":"Boosting Offline Reinforcement Learning via Data Rebalancing","date":"2022-10-17","arxiv_id":"2210.09241","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-predictive-control-via-on-policy","title":"Model Predictive Control via On-Policy Imitation Learning","date":"2022-10-17","arxiv_id":"2210.09206","repositories_listed":0,"syntology":null},{"url":null,"slug":"ptde-personalized-training-with-distillated","title":"PTDE: Personalized Training with Distilled Execution for Multi-Agent Reinforcement Learning","date":"2022-10-17","arxiv_id":"2210.08872","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-only-live-once-single-life-reinforcement","title":"You Only Live Once: Single-Life Reinforcement Learning","date":"2022-10-17","arxiv_id":"2210.08863","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-pipeline-for-offline","title":"Data-Efficient Pipeline for Offline Reinforcement Learning with Limited Data","date":"2022-10-16","arxiv_id":"2210.08642","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-regularized-reinforcement-learning","title":"Entropy Regularized Reinforcement Learning with Cascading Networks","date":"2022-10-16","arxiv_id":"2210.08503","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-task-underspecification-in","title":"The Impact of Task Underspecification in Evaluating Deep Reinforcement Learning","date":"2022-10-16","arxiv_id":"2210.08607","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-an-interpretable-hierarchical-agent","title":"Towards an Interpretable Hierarchical Agent Framework using Semantic Goals","date":"2022-10-16","arxiv_id":"2210.08412","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-scalable-reinforcement-learning-approach","title":"A Scalable Reinforcement Learning Approach for Attack Allocation in Swarm to Swarm Engagement Problems","date":"2022-10-15","arxiv_id":"2210.08319","repositories_listed":0,"syntology":null},{"url":null,"slug":"dyfen-agent-based-fee-setting-in-payment","title":"DyFEn: Agent-Based Fee Setting in Payment Channel Networks","date":"2022-10-15","arxiv_id":"2210.08197","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-regret-bounds-for-multi-batch","title":"Near-Optimal Regret Bounds for Multi-batch Reinforcement Learning","date":"2022-10-15","arxiv_id":"2210.08238","repositories_listed":0,"syntology":null},{"url":null,"slug":"pi-qt-opt-predictive-information-improves","title":"PI-QT-Opt: Predictive Information Improves Multi-Task Robotic Reinforcement Learning at Scale","date":"2022-10-15","arxiv_id":"2210.08217","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-connectx","title":"Reinforcement Learning for ConnectX","date":"2022-10-15","arxiv_id":"2210.08263","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-the-roles-of-text-in-text-games-1","title":"Revisiting the Roles of \"Text\" in Text Games","date":"2022-10-15","arxiv_id":"2210.08384","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-to-3","title":"A Reinforcement Learning Approach to Estimating Long-term Treatment Effects","date":"2022-10-14","arxiv_id":"2210.07536","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-scalable-finite-difference-method-for-deep","title":"A Scalable Finite Difference Method for Deep Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07487","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptable-claim-rewriting-with-offline","title":"Query Rewriting for Effective Misinformation Discovery","date":"2022-10-14","arxiv_id":"2210.07467","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-patch-foraging-in-deep-reinforcement","title":"Adaptive patch foraging in deep reinforcement learning agents","date":"2022-10-14","arxiv_id":"2210.08085","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-trainer-interactive-reinforcement","title":"Multi-trainer Interactive Reinforcement Learning System","date":"2022-10-14","arxiv_id":"2210.08050","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-preference-learning-for-storytelling","title":"Robust Preference Learning for Storytelling via Contrastive Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07792","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-concise-introduction-to-reinforcement","title":"A Concise Introduction to Reinforcement Learning in Robotics","date":"2022-10-13","arxiv_id":"2210.07397","repositories_listed":0,"syntology":null},{"url":null,"slug":"causality-driven-hierarchical-structure","title":"Causality-driven Hierarchical Structure Discovery for Reinforcement Learning","date":"2022-10-13","arxiv_id":"2210.06964","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-automatic-run","title":"Deep reinforcement learning for automatic run-time adaptation of UWB PHY radio settings","date":"2022-10-13","arxiv_id":"2210.15498","repositories_listed":0,"syntology":null},{"url":null,"slug":"dissipative-residual-layers-for-unsupervised","title":"Dissipative residual layers for unsupervised implicit parameterization of data manifolds","date":"2022-10-13","arxiv_id":"2210.07100","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-circuit-implementation-for-coined","title":"Efficient circuit implementation for coined quantum walks on binary trees and application to reinforcement learning","date":"2022-10-13","arxiv_id":"2210.06784","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-category-aware-reinforcement-learning","title":"Object-Category Aware Reinforcement Learning","date":"2022-10-13","arxiv_id":"2210.07802","repositories_listed":0,"syntology":null},{"url":null,"slug":"observed-adversaries-in-deep-reinforcement","title":"Observed Adversaries in Deep Reinforcement Learning","date":"2022-10-13","arxiv_id":"2210.06787","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-control-of-material-micro-structures","title":"Optimal Control of Material Micro-Structures","date":"2022-10-13","arxiv_id":"2210.06734","repositories_listed":0,"syntology":null},{"url":null,"slug":"output-feedback-adaptive-optimal-control-of","title":"Output Feedback Adaptive Optimal Control of Affine Nonlinear systems with a Linear Measurement Model","date":"2022-10-13","arxiv_id":"2210.06637","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-federated-hypernetworks-for","title":"Personalized Federated Hypernetworks for Privacy Preservation in Multi-Task Reinforcement Learning","date":"2022-10-13","arxiv_id":"2210.06820","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-with-serial-markov-chain","title":"Policy Gradient With Serial Markov Chain Reasoning","date":"2022-10-13","arxiv_id":"2210.06766","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-unbiased-policy","title":"Reinforcement Learning with Unbiased Policy Evaluation and Linear Function Approximation","date":"2022-10-13","arxiv_id":"2210.07338","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-multi-agent-reinforcement-learning-2","title":"Towards Multi-Agent Reinforcement Learning driven Over-The-Counter Market Simulations","date":"2022-10-13","arxiv_id":"2210.07184","repositories_listed":0,"syntology":null},{"url":null,"slug":"dqlap-deep-q-learning-recommender-algorithm","title":"DQLAP: Deep Q-Learning Recommender Algorithm with Update Policy for a Real Steam Turbine System","date":"2022-10-12","arxiv_id":"2210.06399","repositories_listed":0,"syntology":null},{"url":null,"slug":"explaining-online-reinforcement-learning","title":"Explaining Online Reinforcement Learning Decisions of Self-Adaptive Systems","date":"2022-10-12","arxiv_id":"2210.05931","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-offline-reinforcement-learning","title":"Real World Offline Reinforcement Learning with Realistic Data Source","date":"2022-10-12","arxiv_id":"2210.06479","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-automated","title":"Reinforcement Learning with Automated Auxiliary Loss Search","date":"2022-10-12","arxiv_id":"2210.06041","repositories_listed":0,"syntology":null},{"url":null,"slug":"smooth-trajectory-collision-avoidance-through","title":"Smooth Trajectory Collision Avoidance through Deep Reinforcement Learning","date":"2022-10-12","arxiv_id":"2210.06377","repositories_listed":0,"syntology":null},{"url":null,"slug":"broad-persistent-advice-for-interactive","title":"Broad-persistent Advice for Interactive Reinforcement Learning Scenarios","date":"2022-10-11","arxiv_id":"2210.05187","repositories_listed":0,"syntology":null},{"url":null,"slug":"edge-cloud-cooperation-for-dnn-inference-via","title":"Edge-Cloud Cooperation for DNN Inference via Reinforcement Learning and Supervised Learning","date":"2022-10-11","arxiv_id":"2210.05182","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-user-reinforcement-learning-with-low","title":"Multi-User Reinforcement Learning with Low Rank Rewards","date":"2022-10-11","arxiv_id":"2210.05355","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-bounds-for-risk-sensitive","title":"Regret Bounds for Risk-Sensitive Reinforcement Learning","date":"2022-10-11","arxiv_id":"2210.05650","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-role-of-exploration-for-task-transfer-in","title":"The Role of Exploration for Task Transfer in Reinforcement Learning","date":"2022-10-11","arxiv_id":"2210.06168","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-a-dynamic-quadrupedal-robotic","title":"Creating a Dynamic Quadrupedal Robotic Goalkeeper with Reinforcement Learning","date":"2022-10-10","arxiv_id":"2210.04435","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-n-step-surrogate-stage-reward-to-reduce","title":"Long N-step Surrogate Stage Reward to Reduce Variances of Deep Reinforcement Learning in Complex Problems","date":"2022-10-10","arxiv_id":"2210.04820","repositories_listed":0,"syntology":null},{"url":null,"slug":"simulating-coverage-path-planning-with-roomba","title":"Simulating Coverage Path Planning with Roomba","date":"2022-10-10","arxiv_id":"2210.04988","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-theoretical-foundation-of-policy","title":"Towards a Theoretical Foundation of Policy Optimization for Learning Control Policies","date":"2022-10-10","arxiv_id":"2210.04810","repositories_listed":0,"syntology":null},{"url":null,"slug":"equivalency-of-optimality-criteria-of-markov","title":"Equivalence of Optimality Criteria for Markov Decision Process and Model Predictive Control","date":"2022-10-09","arxiv_id":"2210.04302","repositories_listed":0,"syntology":null},{"url":"/paper/state-advantage-weighting-for-offline-rl","slug":"state-advantage-weighting-for-offline-rl","title":"State Advantage Weighting for Offline RL","date":"2022-10-09","arxiv_id":"2210.04251","repositories_listed":0,"syntology":{"n":5,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 5 unverified","sample_list":"/paper/state-advantage-weighting-for-offline-rl#ran","syntology_url":"https://syntology.ai/paper/2210.04251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04251"}},"official":null}},{"url":null,"slug":"the-role-of-coverage-in-online-reinforcement","title":"The Role of Coverage in Online Reinforcement Learning","date":"2022-10-09","arxiv_id":"2210.04157","repositories_listed":0,"syntology":null},{"url":null,"slug":"cognitive-models-as-simulators-the-case-of","title":"Cognitive Models as Simulators: The Case of Moral Decision-Making","date":"2022-10-08","arxiv_id":"2210.04121","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamically-meeting-performance-objectives","title":"Dynamically meeting performance objectives for multiple services on a service mesh","date":"2022-10-08","arxiv_id":"2210.04002","repositories_listed":0,"syntology":null},{"url":null,"slug":"advice-conformance-verification-by","title":"Advice Conformance Verification by Reinforcement Learning agents for Human-in-the-Loop","date":"2022-10-07","arxiv_id":"2210.03455","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithmic-trading-using-continuous-action","title":"Algorithmic Trading Using Continuous Action Space Deep Reinforcement Learning","date":"2022-10-07","arxiv_id":"2210.03469","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-enable-uncertainty-estimation-in","title":"How to Enable Uncertainty Estimation in Proximal Policy Optimization","date":"2022-10-07","arxiv_id":"2210.03649","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-context-policy-iteration","title":"Large Language Models can Implement Policy Iteration","date":"2022-10-07","arxiv_id":"2210.03821","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-covering-option-discovery","title":"Multi-agent Deep Covering Skill Discovery","date":"2022-10-07","arxiv_id":"2210.03269","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approach-for-multi","title":"Reinforcement Learning Approach for Multi-Agent Flexible Scheduling Problems","date":"2022-10-07","arxiv_id":"2210.03674","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-inventory-management","title":"Deep Inventory Management","date":"2022-10-06","arxiv_id":"2210.03137","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-human-interactive-recommendation","title":"Digital Human Interactive Recommendation Decision-Making Based on Reinforcement Learning","date":"2022-10-06","arxiv_id":"2210.10638","repositories_listed":0,"syntology":null},{"url":"/paper/distributionally-adaptive-meta-reinforcement","slug":"distributionally-adaptive-meta-reinforcement","title":"Distributionally Adaptive Meta Reinforcement Learning","date":"2022-10-06","arxiv_id":"2210.03104","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/distributionally-adaptive-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.03104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03104"}},"official":null}},{"url":null,"slug":"learning-algorithms-for-intelligent-agents","title":"Learning Algorithms for Intelligent Agents and Mechanisms","date":"2022-10-06","arxiv_id":"2210.02654","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-thrust-orbital-transfer-using-dynamics","title":"Low-Thrust Orbital Transfer using Dynamics-Agnostic Reinforcement Learning","date":"2022-10-06","arxiv_id":"2211.08272","repositories_listed":0,"syntology":null},{"url":null,"slug":"lyapunov-function-consistent-adaptive-network","title":"Lyapunov Function Consistent Adaptive Network Signal Control with Back Pressure and Reinforcement Learning","date":"2022-10-06","arxiv_id":"2210.02612","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-optimal","title":"Meta Reinforcement Learning for Optimal Design of Legged Robots","date":"2022-10-06","arxiv_id":"2210.02750","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-large-action-1","title":"Reinforcement Learning with Large Action Spaces for Neural Machine Translation","date":"2022-10-06","arxiv_id":"2210.03053","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-entropy-maximizing-td3-based","title":"A Novel Entropy-Maximizing TD3-based Reinforcement Learning for Automatic PID Tuning","date":"2022-10-05","arxiv_id":"2210.02381","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-distillation-as-a-state-representation","title":"Neural Distillation as a State Representation Bottleneck in Reinforcement Learning","date":"2022-10-05","arxiv_id":"2210.02224","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-neural-consolidation-for-transfer-in","title":"On Neural Consolidation for Transfer in Reinforcement Learning","date":"2022-10-05","arxiv_id":"2210.02240","repositories_listed":0,"syntology":null},{"url":null,"slug":"query-the-agent-improving-sample-efficiency","title":"Query The Agent: Improving sample efficiency through epistemic uncertainty estimation","date":"2022-10-05","arxiv_id":"2210.02585","repositories_listed":0,"syntology":null}],"record_sha256":"95e8d3899cf01cd6b6cc8b95584e000a0611f340e2f2ac05f6f9a8bb3d6b78e3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}