{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/61","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":61,"pages_in_order":135,"rows_per_page":100,"rows":[6001,6100],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/60","next":"/task/reinforcement-learning-2/papers/62","papers":[{"url":null,"slug":"emergence-of-collective-open-ended","title":"Emergence of Collective Open-Ended Exploration from Decentralized Meta-Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00651","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-natural-policy-gradient-methods-for","title":"Federated Natural Policy Gradient and Actor Critic Methods for Multi-task Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00201","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-impartial-policies-for-sequential","title":"Learning impartial policies for sequential counterfactual explanations using Deep Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00523","repositories_listed":0,"syntology":null},{"url":null,"slug":"mtac-hierarchical-reinforcement-learning","title":"MTAC: Hierarchical Reinforcement Learning-based Multi-gait Terrain-adaptive Quadruped Controller","date":"2023-11-01","arxiv_id":"2401.03337","repositories_listed":0,"syntology":null},{"url":null,"slug":"qfree-a-universal-value-function","title":"QFree: A Universal Value Function Factorization for Multi-Agent Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00356","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-decision-transformer-via","title":"Rethinking Decision Transformer via Hierarchical Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00267","repositories_listed":0,"syntology":null},{"url":null,"slug":"scpo-safe-reinforcement-learning-with-safety","title":"SCPO: Safe Reinforcement Learning with Safety Critic Policy Optimization","date":"2023-11-01","arxiv_id":"2311.00880","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-robotic-reinforcement-learning","title":"Autonomous Robotic Reinforcement Learning with Asynchronous Human Feedback","date":"2023-10-31","arxiv_id":"2310.20608","repositories_listed":0,"syntology":null},{"url":null,"slug":"dropout-strategy-in-reinforcement-learning","title":"Dropout Strategy in Reinforcement Learning: Limiting the Surrogate Objective Variance in Policy Optimization Methods","date":"2023-10-31","arxiv_id":"2310.20380","repositories_listed":0,"syntology":null},{"url":null,"slug":"pick-and-pass-as-a-hat-trick-class-for-first","title":"Closed Drafting as a Case Study for First-Principle Interpretability, Memory, and Generalizability in Deep Reinforcement Learning","date":"2023-10-31","arxiv_id":"2310.20654","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-multi-agent-motion-planning-under","title":"Safe multi-agent motion planning under uncertainty for drones using filtered reinforcement learning","date":"2023-10-31","arxiv_id":"2311.00063","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-aware-causal-representation-for","title":"Safety-aware Causal Representation for Trustworthy Offline Reinforcement Learning in Autonomous Driving","date":"2023-10-31","arxiv_id":"2311.10747","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-alignment-ceiling-objective-mismatch-in","title":"The Alignment Ceiling: Objective Mismatch in Reinforcement Learning from Human Feedback","date":"2023-10-31","arxiv_id":"2311.00168","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-instance-optimality-in-online-pac","title":"Towards Instance-Optimality in Online PAC Reinforcement Learning","date":"2023-10-31","arxiv_id":"2311.05638","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-batch-inverse-reinforcement","title":"Adversarial Batch Inverse Reinforcement Learning: Learn to Reward from Imperfect Demonstration for Interactive Recommendation","date":"2023-10-30","arxiv_id":"2310.19536","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-for-visual-navigation-of","title":"Deep Learning for Visual Navigation of Underwater Robots","date":"2023-10-30","arxiv_id":"2310.19495","repositories_listed":0,"syntology":null},{"url":null,"slug":"dgfn-double-generative-flow-networks","title":"DGFN: Double Generative Flow Networks","date":"2023-10-30","arxiv_id":"2310.19685","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-exploration-in-continuous-time","title":"Efficient Exploration in Continuous-time Model-based Reinforcement Learning","date":"2023-10-30","arxiv_id":"2310.19848","repositories_listed":0,"syntology":null},{"url":null,"slug":"goplan-goal-conditioned-offline-reinforcement","title":"GOPlan: Goal-conditioned Offline Reinforcement Learning by Planning with Learned Models","date":"2023-10-30","arxiv_id":"2310.20025","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-bayesian-regret-bounds-for-thompson","title":"Improved Bayesian Regret Bounds for Thompson Sampling in Reinforcement Learning","date":"2023-10-30","arxiv_id":"2310.20007","repositories_listed":0,"syntology":null},{"url":null,"slug":"remember-what-you-did-so-you-know-what-to-do","title":"Remember what you did so you know what to do next","date":"2023-10-30","arxiv_id":"2311.01468","repositories_listed":0,"syntology":null},{"url":null,"slug":"automaton-distillation-neuro-symbolic","title":"Automaton Distillation: Neuro-Symbolic Transfer Learning for Deep Reinforcement Learning","date":"2023-10-29","arxiv_id":"2310.19137","repositories_listed":0,"syntology":null},{"url":null,"slug":"mag-gnn-reinforcement-learning-boosted-graph","title":"MAG-GNN: Reinforcement Learning Boosted Graph Neural Network","date":"2023-10-29","arxiv_id":"2310.19142","repositories_listed":0,"syntology":null},{"url":null,"slug":"spacecraft-autonomous-decision-planning-for","title":"Spacecraft Autonomous Decision-Planning for Collision Avoidance: a Reinforcement Learning Approach","date":"2023-10-29","arxiv_id":"2310.18966","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-weapons-to","title":"Deep Reinforcement Learning for Weapons to Targets Assignment in a Hypersonic strike","date":"2023-10-27","arxiv_id":"2310.18509","repositories_listed":0,"syntology":null},{"url":null,"slug":"gen2sim-scaling-up-robot-learning-in","title":"Gen2Sim: Scaling up Robot Learning in Simulation with Generative Models","date":"2023-10-27","arxiv_id":"2310.18308","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-data-augmentation-for-offline","title":"Guided Data Augmentation for Offline Reinforcement Learning and Imitation Learning","date":"2023-10-27","arxiv_id":"2310.18247","repositories_listed":0,"syntology":null},{"url":null,"slug":"coalitional-bargaining-via-reinforcement","title":"Coalitional Bargaining via Reinforcement Learning: An Application to Collaborative Vehicle Routing","date":"2023-10-26","arxiv_id":"2310.17458","repositories_listed":0,"syntology":null},{"url":null,"slug":"cqm-curriculum-reinforcement-learning-with-a","title":"CQM: Curriculum Reinforcement Learning with a Quantized World Model","date":"2023-10-26","arxiv_id":"2310.17330","repositories_listed":0,"syntology":null},{"url":null,"slug":"demonstration-regularized-rl","title":"Demonstration-Regularized RL","date":"2023-10-26","arxiv_id":"2310.17303","repositories_listed":0,"syntology":null},{"url":null,"slug":"dsac-c-constrained-maximum-entropy-for-robust","title":"DSAC-C: Constrained Maximum Entropy for Robust Discrete Soft-Actor Critic","date":"2023-10-26","arxiv_id":"2310.17173","repositories_listed":0,"syntology":null},{"url":null,"slug":"fair-collaborative-vehicle-routing-a-deep","title":"Fair collaborative vehicle routing: A deep multi-agent reinforcement learning approach","date":"2023-10-26","arxiv_id":"2310.17485","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphical-object-centric-actor-critic","title":"Relational Object-Centric Actor-Critic","date":"2023-10-26","arxiv_id":"2310.17178","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-as-generalizable","title":"Large Language Models as Generalizable Policies for Embodied Tasks","date":"2023-10-26","arxiv_id":"2310.17722","repositories_listed":0,"syntology":null},{"url":"/paper/reward-scale-robustness-for-proximal-policy","slug":"reward-scale-robustness-for-proximal-policy","title":"Reward Scale Robustness for Proximal Policy Optimization via DreamerV3 Tricks","date":"2023-10-26","arxiv_id":"2310.17805","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-scale-robustness-for-proximal-policy#ran","syntology_url":"https://syntology.ai/paper/2310.17805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17805"}},"official":null}},{"url":null,"slug":"ai-agent-as-urban-planner-steering","title":"AI Agent as Urban Planner: Steering Stakeholder Dynamics in Urban Planning via Consensus-based Multi-Agent Reinforcement Learning","date":"2023-10-25","arxiv_id":"2310.16772","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlled-decoding-from-language-models","title":"Controlled Decoding from Language Models","date":"2023-10-25","arxiv_id":"2310.17022","repositories_listed":0,"syntology":null},{"url":null,"slug":"imperfect-digital-twin-assisted-low-cost","title":"Imperfect Digital Twin Assisted Low Cost Reinforcement Training for Multi-UAV Networks","date":"2023-10-25","arxiv_id":"2310.16302","repositories_listed":0,"syntology":null},{"url":null,"slug":"mimictouch-learning-human-s-control-strategy","title":"MimicTouch: Leveraging Multi-modal Human Tactile Demonstrations for Contact-rich Manipulation","date":"2023-10-25","arxiv_id":"2310.16917","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-enhanced-contrastive-reinforcement","title":"Model-enhanced Contrastive Reinforcement Learning for Sequential Recommendation","date":"2023-10-25","arxiv_id":"2310.16566","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiprompter-cooperative-prompt-optimization","title":"MultiPrompter: Cooperative Prompt Optimization with Multi-Agent Reinforcement Learning","date":"2023-10-25","arxiv_id":"2310.16730","repositories_listed":0,"syntology":null},{"url":null,"slug":"pitfall-of-optimism-distributional","title":"Pitfall of Optimism: Distributional Reinforcement Learning by Randomizing Risk Criterion","date":"2023-10-25","arxiv_id":"2310.16546","repositories_listed":0,"syntology":null},{"url":null,"slug":"privately-aligning-language-models-with","title":"Privately Aligning Language Models with Reinforcement Learning","date":"2023-10-25","arxiv_id":"2310.16960","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-sbm-graphon-games","title":"Reinforcement Learning for SBM Graphon Games with Re-Sampling","date":"2023-10-25","arxiv_id":"2310.16326","repositories_listed":0,"syntology":null},{"url":null,"slug":"symphony-of-experts-orchestration-with","title":"Policy Optimization via Adv2: Adversarial Learning on Advantage Functions","date":"2023-10-25","arxiv_id":"2310.16473","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-control-centric-representations-in","title":"Towards Control-Centric Representations in Reinforcement Learning from Images","date":"2023-10-25","arxiv_id":"2310.16655","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-contextualized-real-time-multimodal-emotion","title":"A Contextualized Real-Time Multimodal Emotion Recognition for Conversational Agents using Graph Convolutional Networks in Reinforcement Learning","date":"2023-10-24","arxiv_id":"2310.18363","repositories_listed":0,"syntology":null},{"url":null,"slug":"copf-continual-learning-human-preference","title":"COPR: Continual Learning Human Preference through Optimal Policy Regularization","date":"2023-10-24","arxiv_id":"2310.15694","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-local-path","title":"Reinforcement learning based local path planning for mobile robot","date":"2023-10-24","arxiv_id":"2403.12463","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-large-flexible-job-shop-scheduling","title":"Solving the flexible job-shop scheduling problem through an enhanced deep reinforcement learning approach","date":"2023-10-24","arxiv_id":"2310.15706","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-doubly-robust-approach-to-sparse","title":"A Doubly Robust Approach to Sparse Reinforcement Learning","date":"2023-10-23","arxiv_id":"2310.15286","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-teacher-selection-for-reinforcement","title":"Active teacher selection for reinforcement learning from human feedback","date":"2023-10-23","arxiv_id":"2310.15288","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-on-the-water-applying-drl-to-autonomous","title":"AI on the Water: Applying DRL to Autonomous Vessel Navigation","date":"2023-10-23","arxiv_id":"2310.14938","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-path-following-in-ships-using","title":"Comparison of path following in ships using modern and traditional controllers","date":"2023-10-23","arxiv_id":"2310.14940","repositories_listed":0,"syntology":null},{"url":null,"slug":"diverse-priors-for-deep-reinforcement","title":"Diverse Priors for Deep Reinforcement Learning","date":"2023-10-23","arxiv_id":"2310.14864","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-robotic-manipulation-harnessing-the","title":"Enhancing Robotic Manipulation: Harnessing the Power of Multi-Task Reinforcement Learning and Single Life Reinforcement Learning in Meta-World","date":"2023-10-23","arxiv_id":"2311.12854","repositories_listed":0,"syntology":null},{"url":null,"slug":"robot-fine-tuning-made-easy-pre-training","title":"Robot Fine-Tuning Made Easy: Pre-Training Rewards and Policies for Autonomous Real-World Reinforcement Learning","date":"2023-10-23","arxiv_id":"2310.15145","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-primacy-bias-in-model-based-rl","title":"Mind the Model, Not the Agent: The Primacy Bias in Model-based RL","date":"2023-10-23","arxiv_id":"2310.15017","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-is-more-diverse-perspectives-within-a","title":"One is More: Diverse Perspectives within a Single Network for Efficient DRL","date":"2023-10-21","arxiv_id":"2310.14009","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-better-match-for-drivers-and-riders","title":"A Better Match for Drivers and Riders: Reinforcement Learning at Lyft","date":"2023-10-20","arxiv_id":"2310.13810","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-multi-agent-deep-reinforcement-2","title":"Cooperative Multi-Agent Deep Reinforcement Learning for Adaptive Decentralized Emergency Voltage Control","date":"2023-10-20","arxiv_id":"2310.13577","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-deep-reinforcement-learning-for-1","title":"Interpretable Deep Reinforcement Learning for Optimizing Heterogeneous Energy Storage Systems","date":"2023-10-20","arxiv_id":"2310.14783","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-shaping-for-happier-autonomous-cyber","title":"Reward Shaping for Happier Autonomous Cyber Security Agents","date":"2023-10-20","arxiv_id":"2310.13565","repositories_listed":0,"syntology":null},{"url":null,"slug":"tree-search-in-dag-space-with-model-based","title":"Tree Search in DAG Space with Model-based Reinforcement Learning for Causal Discovery","date":"2023-10-20","arxiv_id":"2310.13576","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-gymnasium-a-unified-safe-reinforcement","title":"Safety-Gymnasium: A Unified Safe Reinforcement Learning Benchmark","date":"2023-10-19","arxiv_id":"2310.12567","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-pac-learning-algorithm-for-ltl-and-omega","title":"A PAC Learning Algorithm for LTL and Omega-regular Objectives in MDPs","date":"2023-10-18","arxiv_id":"2310.12248","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerate-presolve-in-large-scale-linear","title":"Accelerate Presolve in Large-Scale Linear Programming via Reinforcement Learning","date":"2023-10-18","arxiv_id":"2310.11845","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-quantized-offline-reinforcement","title":"Action-Quantized Offline Reinforcement Learning for Robotic Skill Learning","date":"2023-10-18","arxiv_id":"2310.11731","repositories_listed":0,"syntology":null},{"url":null,"slug":"fact-based-agent-modeling-for-multi-agent","title":"Fact-based Agent modeling for Multi-Agent Reinforcement Learning","date":"2023-10-18","arxiv_id":"2310.12290","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-generalization-of-alignment-with","title":"Improving Generalization of Alignment with Human Preferences through Group Invariant Learning","date":"2023-10-18","arxiv_id":"2310.11971","repositories_listed":0,"syntology":null},{"url":null,"slug":"marvel-multi-agent-reinforcement-learning-for","title":"MARVEL: Multi-Agent Reinforcement-Learning for Large-Scale Variable Speed Limits","date":"2023-10-18","arxiv_id":"2310.12359","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-expressivity-of-objective","title":"On The Expressivity of Objective-Specification Formalisms in Reinforcement Learning","date":"2023-10-18","arxiv_id":"2310.11840","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-acceleration-of-infinite-horizon","title":"Quantum Speedups in Regret Analysis of Infinite Horizon Average-Reward Markov Decision Processes","date":"2023-10-18","arxiv_id":"2310.11684","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-reward-ambiguity-through","title":"Understanding Reward Ambiguity Through Optimal Transport Theory in Inverse Reinforcement Learning","date":"2023-10-18","arxiv_id":"2310.12055","repositories_listed":0,"syntology":null},{"url":null,"slug":"combat-urban-congestion-via-collaboration","title":"Combat Urban Congestion via Collaboration: Heterogeneous GNN-based MARL for Coordinated Platooning and Traffic Signal Control","date":"2023-10-17","arxiv_id":"2310.10948","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-packing-from-visual-sensing-to","title":"Neural Packing: from Visual Sensing to Reinforcement Learning","date":"2023-10-17","arxiv_id":"2311.09233","repositories_listed":0,"syntology":null},{"url":null,"slug":"reaching-the-limit-in-autonomous-racing","title":"Reaching the Limit in Autonomous Racing: Optimal Control versus Reinforcement Learning","date":"2023-10-17","arxiv_id":"2310.10943","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-transfer-of-adaptive-control","title":"Sim-to-Real Transfer of Adaptive Control Parameters for AUV Stabilization under Current Disturbance","date":"2023-10-17","arxiv_id":"2310.11075","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-offline-reinforcement-learning-for","title":"End-to-end Offline Reinforcement Learning for Glycemia Control","date":"2023-10-16","arxiv_id":"2310.10312","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-topological-maps-in-deep","title":"Leveraging Topological Maps in Deep Reinforcement Learning for Multi-Object Navigation","date":"2023-10-16","arxiv_id":"2310.10250","repositories_listed":0,"syntology":null},{"url":null,"slug":"mimicking-the-maestro-exploring-the-efficacy","title":"Mimicking the Maestro: Exploring the Efficacy of a Virtual AI Teacher in Fine Motor Skill Acquisition","date":"2023-10-16","arxiv_id":"2310.10280","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-complexity-of-preference-based","title":"Sample Complexity of Preference-Based Nonparametric Off-Policy Evaluation with Deep Networks","date":"2023-10-16","arxiv_id":"2310.10556","repositories_listed":0,"syntology":null},{"url":null,"slug":"verbosity-bias-in-preference-labeling-by","title":"Verbosity Bias in Preference Labeling by Large Language Models","date":"2023-10-16","arxiv_id":"2310.10076","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-explicit","title":"Deep Reinforcement Learning with Explicit Context Representation","date":"2023-10-15","arxiv_id":"2310.09924","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-reinforcement-learning-for-resource","title":"Federated Reinforcement Learning for Resource Allocation in V2X Networks","date":"2023-10-15","arxiv_id":"2310.09858","repositories_listed":0,"syntology":null},{"url":null,"slug":"mir2-towards-provably-robust-multi-agent","title":"Robust Multi-Agent Reinforcement Learning by Mutual Information Regularization","date":"2023-10-15","arxiv_id":"2310.09833","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-blockchain-empowered-multi-aggregator","title":"A Blockchain-empowered Multi-Aggregator Federated Learning Architecture in Edge Computing with Deep Reinforcement Learning Optimization","date":"2023-10-14","arxiv_id":"2310.09665","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-empowering-reinforcement","title":"A Framework for Empowering Reinforcement Learning Agents with Causal Analysis: Enhancing Automated Cryptocurrency Trading","date":"2023-10-14","arxiv_id":"2310.09462","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-music-playlist-generation-via","title":"Automatic Music Playlist Generation via Simulation-based Reinforcement Learning","date":"2023-10-13","arxiv_id":"2310.09123","repositories_listed":0,"syntology":null},{"url":null,"slug":"community-membership-hiding-as-counterfactual","title":"Evading Community Detection via Counterfactual Neighborhood Search","date":"2023-10-13","arxiv_id":"2310.08909","repositories_listed":0,"syntology":null},{"url":null,"slug":"goodhart-s-law-in-reinforcement-learning","title":"Goodhart's Law in Reinforcement Learning","date":"2023-10-13","arxiv_id":"2310.09144","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-for-optimizing","title":"Offline Reinforcement Learning for Optimizing Production Bidding Policies","date":"2023-10-13","arxiv_id":"2310.09426","repositories_listed":0,"syntology":null},{"url":null,"slug":"urban-drone-navigation-autoencoder-learning","title":"Urban Drone Navigation: Autoencoder Learning Fusion for Aerodynamics","date":"2023-10-13","arxiv_id":"2310.08830","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-way-to-incorporate-novelty-detection","title":"Novelty Detection in Reinforcement Learning with World Models","date":"2023-10-12","arxiv_id":"2310.08731","repositories_listed":0,"syntology":null},{"url":null,"slug":"meanap-guided-reinforced-active-learning-for","title":"Aligning Data Selection with Performance: Performance-driven Reinforcement Learning for Active Learning in Object Detection","date":"2023-10-12","arxiv_id":"2310.08387","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-display-transfer","title":"Reinforcement Learning of Display Transfer Robots in Glass Flow Control Systems: A Physical Simulation-Based Approach","date":"2023-10-12","arxiv_id":"2310.07981","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-autonomous-5","title":"Deep Reinforcement Learning for Autonomous Cyber Defence: A Survey","date":"2023-10-11","arxiv_id":"2310.07745","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepref-deep-reinforcement-learning-for-video","title":"DeePref: Deep Reinforcement Learning For Video Prefetching In Content Delivery Networks","date":"2023-10-11","arxiv_id":"2310.07881","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-causal-graph-priors-with-posterior","title":"Exploiting Causal Graph Priors with Posterior Sampling for Reinforcement Learning","date":"2023-10-11","arxiv_id":"2310.07518","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-knowledge-graph","title":"Reinforcement Learning-based Knowledge Graph Reasoning for Explainable Fact-checking","date":"2023-10-11","arxiv_id":"2310.07613","repositories_listed":0,"syntology":null}],"record_sha256":"7f41b5b28fa1ade6b88c07fb48fbd4e07c01d0a0da5f5cd28951328b202ecaa1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}