{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/80","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":80,"pages_in_order":152,"rows_per_page":100,"rows":[7901,8000],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/79","next":"/task/reinforcement-learning-1/papers/81","papers":[{"url":null,"slug":"visual-backtracking-teleoperation-a-data","title":"Visual Backtracking Teleoperation: A Data Collection Protocol for Offline Image-Based Reinforcement Learning","date":"2022-10-05","arxiv_id":"2210.02343","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-scheduling","title":"Using Deep Reinforcement Learning for mmWave Real-Time Scheduling","date":"2022-10-04","arxiv_id":"2210.01423","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-disentanglement-in-generative","title":"Evaluating Disentanglement in Generative Models Without Knowledge of Latent Factors","date":"2022-10-04","arxiv_id":"2210.01760","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-reinforcement-learning-for-real","title":"Federated Reinforcement Learning for Real-Time Electric Vehicle Charging and Discharging Control","date":"2022-10-04","arxiv_id":"2210.01452","repositories_listed":0,"syntology":null},{"url":null,"slug":"handling-sparse-rewards-in-reinforcement","title":"Handling Sparse Rewards in Reinforcement Learning Using Model Predictive Control","date":"2022-10-04","arxiv_id":"2210.01525","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperbolic-deep-reinforcement-learning","title":"Hyperbolic Deep Reinforcement Learning","date":"2022-10-04","arxiv_id":"2210.01542","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dynamic-abstract-representations-for","title":"Learning Dynamic Abstract Representations for Sample-Efficient Reinforcement Learning","date":"2022-10-04","arxiv_id":"2210.01955","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-perception-aware-agile-flight-in","title":"Learning Perception-Aware Agile Flight in Cluttered Environments","date":"2022-10-04","arxiv_id":"2210.01841","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-likelihood-inverse-reinforcement","title":"Maximum-Likelihood Inverse Reinforcement Learning with Finite-Time Guarantees","date":"2022-10-04","arxiv_id":"2210.01808","repositories_listed":0,"syntology":null},{"url":null,"slug":"costnet-an-end-to-end-framework-for-goal","title":"CostNet: An End-to-End Framework for Goal-Directed Reinforcement Learning","date":"2022-10-03","arxiv_id":"2210.01805","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-option-discovery-using-deep-q","title":"Interpretable Option Discovery using Deep Q-Learning and Variational Autoencoders","date":"2022-10-03","arxiv_id":"2210.01231","repositories_listed":0,"syntology":null},{"url":null,"slug":"mastering-spatial-graph-prediction-of-road","title":"Mastering Spatial Graph Prediction of Road Networks","date":"2022-10-03","arxiv_id":"2210.00828","repositories_listed":0,"syntology":null},{"url":null,"slug":"msrl-distributed-reinforcement-learning-with","title":"MSRL: Distributed Reinforcement Learning with Dataflow Fragments","date":"2022-10-03","arxiv_id":"2210.00882","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-deployment-efficiency-in-reward","title":"Near-Optimal Deployment Efficiency in Reward-Free Reinforcement Learning with Linear Function Approximation","date":"2022-10-03","arxiv_id":"2210.00701","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-5","title":"Offline Reinforcement Learning with Differentiable Function Approximation is Provably Efficient","date":"2022-10-03","arxiv_id":"2210.00750","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-for-reinforcement-learning","title":"Policy Gradient for Reinforcement Learning with General Utilities","date":"2022-10-03","arxiv_id":"2210.00991","repositories_listed":0,"syntology":null},{"url":null,"slug":"square-root-regret-bounds-for-continuous-time","title":"Square-root regret bounds for continuous-time episodic Markov decision processes","date":"2022-10-03","arxiv_id":"2210.00832","repositories_listed":0,"syntology":null},{"url":null,"slug":"euclid-towards-efficient-unsupervised","title":"EUCLID: Towards Efficient Unsupervised Reinforcement Learning with Multi-choice Dynamics Model","date":"2022-10-02","arxiv_id":"2210.00498","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradients-for-probabilistic","title":"Policy Gradients for Probabilistic Constrained Reinforcement Learning","date":"2022-10-02","arxiv_id":"2210.00596","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-bayesian-optimization-with","title":"Robust Bayesian optimization with reinforcement learned acquisition functions","date":"2022-10-02","arxiv_id":"2210.00476","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-q-learning-with-imperfect-expert","title":"Bayesian Q-learning With Imperfect Expert Demonstrations","date":"2022-10-01","arxiv_id":"2210.01800","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-data-diversity-enhance-learning","title":"Can Data Diversity Enhance Learning Generalization?","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-bert-based-reward-functions-for","title":"Comparing BERT-based Reward Functions for Deep Reinforcement Learning in Machine Translation","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-adaptive-optimal-control-of","title":"Learning-Based Adaptive Optimal Control of Linear Time-Delay Systems: A Policy Iteration Approach","date":"2022-10-01","arxiv_id":"2210.00204","repositories_listed":0,"syntology":null},{"url":null,"slug":"parsing-natural-language-into-propositional-1","title":"Parsing Natural Language into Propositional and First-Order Logic with Dual Reinforcement Learning","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-policy-transfer-with-disentangled-1","title":"Zero-Shot Policy Transfer with Disentangled Task Representation of Meta-Reinforcement Learning","date":"2022-10-01","arxiv_id":"2210.00350","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-general-framework-for-sample-efficient","title":"A General Framework for Sample-Efficient Function Approximation in Reinforcement Learning","date":"2022-09-30","arxiv_id":"2209.15634","repositories_listed":0,"syntology":null},{"url":null,"slug":"aspire-adaptive-skill-priors-for","title":"ASPiRe:Adaptive Skill Priors for Reinforcement Learning","date":"2022-09-30","arxiv_id":"2209.15205","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-lstm-training-with-eligibility","title":"Efficient LSTM Training with Eligibility Traces","date":"2022-09-30","arxiv_id":"2209.15502","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficiently-learning-small-policies-for","title":"Efficiently Learning Small Policies for Locomotion and Manipulation","date":"2022-09-30","arxiv_id":"2210.00140","repositories_listed":0,"syntology":null},{"url":null,"slug":"observational-robustness-and-invariances-in","title":"Bounded Robustness in Reinforcement Learning via Lexicographic Objectives","date":"2022-09-30","arxiv_id":"2209.15320","repositories_listed":0,"syntology":null},{"url":null,"slug":"programmable-control-of-ultrasound-swarmbots","title":"Programmable Control of Ultrasound Swarmbots through Reinforcement Learning","date":"2022-09-30","arxiv_id":"2209.15393","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-shaping-for-user-satisfaction-in-a","title":"Reward Shaping for User Satisfaction in a REINFORCE Recommender","date":"2022-09-30","arxiv_id":"2209.15166","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-md-a-novel-reinforcement-learning-approach","title":"RL-MD: A Novel Reinforcement Learning Approach for DNA Motif Discovery","date":"2022-09-30","arxiv_id":"2209.15181","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-role-of-time-delay-in-sim2real-transfer","title":"The Role of Time Delay in Sim2real Transfer of Reinforcement Learning for Cyber-Physical Systems","date":"2022-09-30","arxiv_id":"2209.15216","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-fully-autonomous-uav-controller-for","title":"Towards a Fully Autonomous UAV Controller for Moving Platform Detection and Landing","date":"2022-09-30","arxiv_id":"2210.08120","repositories_listed":0,"syntology":null},{"url":null,"slug":"blessing-from-experts-super-reinforcement","title":"Blessing from Human-AI Interaction: Super Reinforcement Learning in Confounded Environments","date":"2022-09-29","arxiv_id":"2209.15448","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-unsupervised-learning-of-world","title":"Contrastive Unsupervised Learning of World Model with Invariant Causal Features","date":"2022-09-29","arxiv_id":"2209.14932","repositories_listed":0,"syntology":null},{"url":null,"slug":"enforcing-hard-constraints-with-soft-barriers","title":"Enforcing Hard Constraints with Soft Barriers: Safe Reinforcement Learning in Unknown Stochastic Environments","date":"2022-09-29","arxiv_id":"2209.15090","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-training-of-deep-ensemble","title":"Ensemble Reinforcement Learning in Continuous Spaces -- A Hierarchical Multi-Step Approach for Policy Training","date":"2022-09-29","arxiv_id":"2209.14488","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-does-value-distribution-in-distributional","title":"How Does Return Distribution in Distributional Reinforcement Learning Help Optimization?","date":"2022-09-29","arxiv_id":"2209.14513","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-parsimonious-dynamics-for","title":"Learning Parsimonious Dynamics for Generalization in Reinforcement Learning","date":"2022-09-29","arxiv_id":"2209.14781","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-weighted-q-ensembles-for-reduced","title":"Online Weighted Q-Ensembles for Reduced Hyperparameter Tuning in Reinforcement Learning","date":"2022-09-29","arxiv_id":"2209.15078","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimistic-mle-a-generic-model-based","title":"Optimistic MLE -- A Generic Model-based Algorithm for Partially Observable Sequential Decision Making","date":"2022-09-29","arxiv_id":"2209.14997","repositories_listed":0,"syntology":null},{"url":null,"slug":"partially-observable-rl-with-b-stability","title":"Partially Observable RL with B-Stability: Unified Structural Condition and Sharp Sample-Efficient Algorithms","date":"2022-09-29","arxiv_id":"2209.14990","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-algorithms-an-overview","title":"Reinforcement Learning Algorithms: An Overview and Classification","date":"2022-09-29","arxiv_id":"2209.14940","repositories_listed":0,"syntology":null},{"url":null,"slug":"argumentative-reward-learning-reasoning-about","title":"Argumentative Reward Learning: Reasoning About Human Preferences","date":"2022-09-28","arxiv_id":"2209.14010","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-transfer-in-continual","title":"Disentangling Transfer in Continual Reinforcement Learning","date":"2022-09-28","arxiv_id":"2209.13900","repositories_listed":0,"syntology":null},{"url":null,"slug":"fire-a-failure-adaptive-reinforcement","title":"FIRE: A Failure-Adaptive Reinforcement Learning Framework for Edge Computing Migrations","date":"2022-09-28","arxiv_id":"2209.14399","repositories_listed":0,"syntology":null},{"url":null,"slug":"guiding-safe-exploration-with-weakest","title":"Guiding Safe Exploration with Weakest Preconditions","date":"2022-09-28","arxiv_id":"2209.14148","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-generalization-of-deep-reinforcement","title":"Generalization in Deep Reinforcement Learning for Robotic Navigation by Reward Shaping","date":"2022-09-28","arxiv_id":"2209.14271","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-policy-optimization-for-robust-mdp","title":"Online Policy Optimization for Robust MDP","date":"2022-09-28","arxiv_id":"2209.13841","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-crypto-asset-automated-market","title":"Predictive Crypto-Asset Automated Market Making Architecture for Decentralized Finance using Deep Reinforcement Learning","date":"2022-09-28","arxiv_id":"2211.01346","repositories_listed":0,"syntology":null},{"url":null,"slug":"dce-offline-reinforcement-learning-with","title":"DCE: Offline Reinforcement Learning With Double Conservative Estimates","date":"2022-09-27","arxiv_id":"2209.13132","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-of-experiments-for-the-calibration-of","title":"Design of experiments for the calibration of history-dependent models via deep reinforcement learning and an enhanced Kalman filter","date":"2022-09-27","arxiv_id":"2209.13126","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-frank-wolfe-policy-optimization-for","title":"Neural Frank-Wolfe Policy Optimization for Region-of-Interest Intra-Frame Coding with HEVC/H.265","date":"2022-09-27","arxiv_id":"2209.13210","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-cognitive-delay","title":"Reinforcement Learning for Cognitive Delay/Disruption Tolerant Network Node Management in an LEO-based Satellite Constellation","date":"2022-09-27","arxiv_id":"2209.13237","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-non-exponential","title":"Reinforcement Learning with Non-Exponential Discounting","date":"2022-09-27","arxiv_id":"2209.13413","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-of-dynamic-high","title":"Safe Reinforcement Learning of Dynamic High-Dimensional Robotic Tasks: Navigation, Manipulation, Interaction","date":"2022-09-27","arxiv_id":"2209.13308","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-network-for-o-ran-resource","title":"Actor-Critic Network for O-RAN Resource Allocation: xApp Design, Deployment, and Analysis","date":"2022-09-26","arxiv_id":"2210.04604","repositories_listed":0,"syntology":null},{"url":null,"slug":"deft-diverse-ensembles-for-fast-transfer-in","title":"DEFT: Diverse Ensembles for Fast Transfer in Reinforcement Learning","date":"2022-09-26","arxiv_id":"2209.12412","repositories_listed":0,"syntology":null},{"url":null,"slug":"delayed-geometric-discounts-an-alternative-1","title":"Delayed Geometric Discounts: An Alternative Criterion for Reinforcement Learning","date":"2022-09-26","arxiv_id":"2209.12483","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-document-image-understanding-with","title":"Improving Document Image Understanding with Reinforcement Finetuning","date":"2022-09-26","arxiv_id":"2209.12561","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-referential-ambiguity-in-language","title":"Overcoming Referential Ambiguity in Language-Guided Goal-Conditioned Reinforcement Learning","date":"2022-09-26","arxiv_id":"2209.12758","repositories_listed":0,"syntology":null},{"url":null,"slug":"paused-agent-replay-refresh","title":"Paused Agent Replay Refresh","date":"2022-09-26","arxiv_id":"2209.13398","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-hindsight-goal-relabeling","title":"Understanding Hindsight Goal Relabeling from a Divergence Minimization Perspective","date":"2022-09-26","arxiv_id":"2209.13046","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-adaptive-mesh","title":"Deep Reinforcement Learning for Adaptive Mesh Refinement","date":"2022-09-25","arxiv_id":"2209.12351","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-opportunities-and-challenges-of-using","title":"Opportunities and Challenges from Using Animal Videos in Reinforcement Learning for Navigation","date":"2022-09-25","arxiv_id":"2209.12347","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-learning-using-structural-motifs-in","title":"Reward Learning using Structural Motifs in Inverse Reinforcement Learning","date":"2022-09-25","arxiv_id":"2209.13489","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-lifelong-adaptive-inverse-reinforcement","title":"Fast Lifelong Adaptive Inverse Reinforcement Learning from Demonstrations","date":"2022-09-24","arxiv_id":"2209.11908","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantification-before-selection-active","title":"Quantification before Selection: Active Dynamics Preference for Robust Reinforcement Learning","date":"2022-09-23","arxiv_id":"2209.11596","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-real-world-reinforcement-learning-for","title":"SAFER: Safe Collision Avoidance using Focused and Efficient Trajectory Search with Reinforcement Learning","date":"2022-09-23","arxiv_id":"2209.11789","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-algorithms-for-rl-with-decision","title":"Unified Algorithms for RL with Decision-Estimation Coefficients: PAC, Reward-Free, Preference-Based Learning, and Beyond","date":"2022-09-23","arxiv_id":"2209.11745","repositories_listed":0,"syntology":null},{"url":null,"slug":"computational-discovery-of-energy-efficient","title":"Computational Discovery of Energy-Efficient Heat Treatment for Microstructure Design using Deep Reinforcement Learning","date":"2022-09-22","arxiv_id":"2209.11259","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-evaluating-and-scaling-learning","title":"Developing, Evaluating and Scaling Learning Agents in Multi-Agent Environments","date":"2022-09-22","arxiv_id":"2209.10958","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimizing-human-assistance-augmenting-a","title":"Minimizing Human Assistance: Augmenting a Single Demonstration for Deep Reinforcement Learning","date":"2022-09-22","arxiv_id":"2209.11275","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-reinforcement-learning-simulation","title":"Parallel Reinforcement Learning Simulation for Visual Quadrotor Navigation","date":"2022-09-22","arxiv_id":"2209.11094","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-computing-and","title":"Reinforcement Learning in Computing and Network Convergence Orchestration","date":"2022-09-22","arxiv_id":"2209.10753","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-framework-with","title":"ECSAS: Exploring Critical Scenarios from Action Sequence in Autonomous Driving","date":"2022-09-21","arxiv_id":"2209.10078","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-look-ahead-economic-dispatch","title":"Evaluation of Look-ahead Economic Dispatch Using Reinforcement Learning","date":"2022-09-21","arxiv_id":"2209.10207","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-decision-transformer","title":"Hierarchical Decision Transformer","date":"2022-09-21","arxiv_id":"2209.10447","repositories_listed":0,"syntology":null},{"url":null,"slug":"lamarckian-platform-pushing-the-boundaries-of","title":"Lamarckian Platform: Pushing the Boundaries of Evolutionary Reinforcement Learning towards Asynchronous Commercial Games","date":"2022-09-21","arxiv_id":"2209.10055","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-symmetry-meta-reinforcement","title":"Learning from Symmetry: Meta-Reinforcement Learning with Symmetrical Behaviors and Language Instructions","date":"2022-09-21","arxiv_id":"2209.10656","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-reinforcement-learning-for-asset","title":"Model-Free Reinforcement Learning for Asset Allocation","date":"2022-09-21","arxiv_id":"2209.10458","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-convergence-theory-of-meta","title":"On the Convergence Theory of Meta Reinforcement Learning with Personalized Policies","date":"2022-09-21","arxiv_id":"2209.10072","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-optimization-for-variable","title":"Performance Optimization for Variable Bitwidth Federated Learning in Wireless Networks","date":"2022-09-21","arxiv_id":"2209.10200","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-based-charging","title":"A Deep Reinforcement Learning-Based Charging Scheduling Approach with Augmented Lagrangian for Electric Vehicle","date":"2022-09-20","arxiv_id":"2209.09772","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-spiking-neural-network-learning-markov","title":"A Spiking Neural Network Learning Markov Chain","date":"2022-09-20","arxiv_id":"2209.09572","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-actor-critic-for-multi-agent","title":"Asynchronous Actor-Critic for Multi-Agent Reinforcement Learning","date":"2022-09-20","arxiv_id":"2209.10113","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-network-for-ai-soccer","title":"Deep Q-Network for AI Soccer","date":"2022-09-20","arxiv_id":"2209.09491","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-value-iteration","title":"Graph Value Iteration","date":"2022-09-20","arxiv_id":"2209.09608","repositories_listed":0,"syntology":null},{"url":null,"slug":"irs-assisted-noma-aided-mobile-edge-computing","title":"IRS Assisted NOMA Aided Mobile Edge Computing with Queue Stability: Heterogeneous Multi-Agent Reinforcement Learning","date":"2022-09-20","arxiv_id":"2209.09776","repositories_listed":0,"syntology":null},{"url":null,"slug":"locally-constrained-representations-in","title":"Locally Constrained Representations in Reinforcement Learning","date":"2022-09-20","arxiv_id":"2209.09441","repositories_listed":0,"syntology":null},{"url":null,"slug":"macro-action-based-multi-agent-robot-deep","title":"Macro-Action-Based Multi-Agent/Robot Deep Reinforcement Learning under Partial Observability","date":"2022-09-20","arxiv_id":"2209.10003","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-crop-management-with-reinforcement","title":"Optimizing Crop Management with Reinforcement Learning and Imitation Learning","date":"2022-09-20","arxiv_id":"2209.09991","repositories_listed":0,"syntology":null},{"url":null,"slug":"soft-action-priors-towards-robust-policy","title":"Soft Action Priors: Towards Robust Policy Transfer","date":"2022-09-20","arxiv_id":"2209.09882","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-task-prioritized-policy-composition","title":"Towards Task-Prioritized Policy Composition","date":"2022-09-20","arxiv_id":"2209.09536","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-transferable-and-automatic-tuning-of-deep","title":"A Transferable and Automatic Tuning of Deep Reinforcement Learning for Cost Effective Phishing Detection","date":"2022-09-19","arxiv_id":"2209.09033","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-predicting-coding-brain-inspired","title":"Active Predicting Coding: Brain-Inspired Reinforcement Learning for Sparse Reward Robotic Control Problems","date":"2022-09-19","arxiv_id":"2209.09174","repositories_listed":0,"syntology":null},{"url":null,"slug":"age-of-semantics-in-cooperative","title":"Age of Semantics in Cooperative Communications: To Expedite Simulation Towards Real via Offline Reinforcement Learning","date":"2022-09-19","arxiv_id":"2209.08947","repositories_listed":0,"syntology":null}],"record_sha256":"16c6640a62de63bcee468ed9a4f67dcda14da2e98339839ef131ced16649ffa3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}