{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/continuous-control/papers/9","list_of":"/task/continuous-control","task":"Continuous Control","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":12,"rows_per_page":100,"rows":[801,900],"of":1161,"counts":{"archive_papers_tagged":1161,"with_a_code_link":494,"where_syntology_ran_a_sample":193,"not_listed_spam_title":0,"listed":1161,"listed_where_code_ran":193,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":32,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":32,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/continuous-control","prev":"/task/continuous-control/papers/8","next":"/task/continuous-control/papers/10","papers":[{"url":null,"slug":"wish-you-were-here-hindsight-goal-selection-1","title":"Wish you were here: Hindsight Goal Selection for long-horizon dexterous manipulation","date":"2021-12-01","arxiv_id":"2112.00597","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-low-rank-q-matrix","title":"Uncertainty-aware Low-Rank Q-Matrix Estimation for Deep Reinforcement Learning","date":"2021-11-19","arxiv_id":"2111.10103","repositories_listed":0,"syntology":null},{"url":null,"slug":"aggressive-q-learning-with-ensembles-1","title":"Aggressive Q-Learning with Ensembles: Achieving Both High Sample Efficiency and High Asymptotic Performance","date":"2021-11-17","arxiv_id":"2111.09159","repositories_listed":0,"syntology":null},{"url":"/paper/gri-general-reinforced-imitation-and-its","slug":"gri-general-reinforced-imitation-and-its","title":"GRI: General Reinforced Imitation and its Application to Vision-Based Autonomous Driving","date":"2021-11-16","arxiv_id":"2111.08575","repositories_listed":0,"syntology":null},{"url":null,"slug":"obstacle-avoidance-for-uas-in-continuous","title":"Obstacle Avoidance for UAS in Continuous Action Space Using Deep Reinforcement Learning","date":"2021-11-13","arxiv_id":"2111.07037","repositories_listed":0,"syntology":null},{"url":null,"slug":"awd3-dynamic-reduction-of-the-estimation-bias","title":"AWD3: Dynamic Reduction of the Estimation Bias","date":"2021-11-12","arxiv_id":"2111.06780","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-bang-bang-control-all-you-need-solving","title":"Is Bang-Bang Control All You Need? Solving Continuous Control with Bernoulli Policies","date":"2021-11-03","arxiv_id":"2111.02552","repositories_listed":0,"syntology":null},{"url":null,"slug":"proximal-policy-optimization-with-continuous","title":"Proximal Policy Optimization with Continuous Bounded Action Space via the Beta Distribution","date":"2021-11-03","arxiv_id":"2111.02202","repositories_listed":0,"syntology":null},{"url":null,"slug":"smooth-imitation-learning-via-smooth-costs","title":"Smooth Imitation Learning via Smooth Costs and Smooth Policies","date":"2021-11-03","arxiv_id":"2111.02354","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-correction-for-deep-deterministic","title":"Off-Policy Correction for Deep Deterministic Policy Gradient Algorithms via Batch Prioritized Experience Replay","date":"2021-11-02","arxiv_id":"2111.01865","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-dynamic-bus-control-a-distributional","title":"Robust Dynamic Bus Control: A Distributional Multi-agent Reinforcement Learning Approach","date":"2021-11-02","arxiv_id":"2111.01946","repositories_listed":0,"syntology":null},{"url":null,"slug":"adjacency-constraint-for-efficient","title":"Adjacency constraint for efficient hierarchical reinforcement learning","date":"2021-10-30","arxiv_id":"2111.00213","repositories_listed":0,"syntology":null},{"url":null,"slug":"dream-to-explore-adaptive-simulations-for","title":"Dream to Explore: Adaptive Simulations for Autonomous Systems","date":"2021-10-27","arxiv_id":"2110.14157","repositories_listed":0,"syntology":null},{"url":null,"slug":"automating-control-of-overestimation-bias-for","title":"Automating Control of Overestimation Bias for Reinforcement Learning","date":"2021-10-26","arxiv_id":"2110.13523","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-distributed-actor-critic-architecture","title":"Fully Distributed Actor-Critic Architecture for Multitask Deep Reinforcement Learning","date":"2021-10-23","arxiv_id":"2110.12306","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-reinforcement-learning-with-2","title":"Off-policy Reinforcement Learning with Optimistic Exploration and Distribution Correction","date":"2021-10-22","arxiv_id":"2110.12081","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-high-variance-unavoidable-in-rl-a-case-1","title":"Is High Variance Unavoidable in RL? A Case Study in Continuous Control","date":"2021-10-21","arxiv_id":"2110.11222","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-dynamics-inverse-reinforcement-learning","title":"Off-Dynamics Inverse Reinforcement Learning from Hetero-Domain","date":"2021-10-21","arxiv_id":"2110.11443","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-soft","title":"Offline Reinforcement Learning with Soft Behavior Regularization","date":"2021-10-14","arxiv_id":"2110.07395","repositories_listed":0,"syntology":null},{"url":null,"slug":"braxlines-fast-and-interactive-toolkit-for-rl","title":"Braxlines: Fast and Interactive Toolkit for RL-driven Behavior Engineering beyond Reward Maximization","date":"2021-10-10","arxiv_id":"2110.04686","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-model-based-planning-and-planner","title":"Evaluating model-based planning and planner amortization for continuous control","date":"2021-10-07","arxiv_id":"2110.03363","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-pessimism-for-robust-and-efficient","title":"Learning Pessimism for Robust and Efficient Off-Policy Reinforcement Learning","date":"2021-10-07","arxiv_id":"2110.03375","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experimental-design-perspective-on-1","title":"An Experimental Design Perspective on Exploration in Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-cross-entropy-method-for-model","title":"Decentralized Cross-Entropy Method for Model-Based Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-decision-transformer-for","title":"Distributional Decision Transformer for Hindsight Information Matching","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"effects-of-conservatism-on-offline-learning","title":"Effects of Conservatism on Offline Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-robustness-of-cooperative-marl","title":"Evaluating Robustness of Cooperative MARL","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-diversity-optimization-with","title":"Evolutionary Diversity Optimization with Clustering-based Selection for Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"faster-reinforcement-learning-with-value","title":"Faster Reinforcement Learning with Value Target Lower Bounding","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fight-fire-with-fire-countering-bad-shortcuts","title":"Fight fire with fire: countering bad shortcuts in imitation learning with good shortcuts","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizing-successor-features-to-continuous","title":"Generalizing Successor Features to continuous domains for Multi-task Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-enhanced-exploration-for-goal-oriented","title":"Graph-Enhanced Exploration for Goal-oriented Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-learning-from-pixel-observations","title":"Imitation Learning from Pixel Observations for Continuous Control","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-safety-in-deep-reinforcement","title":"Improving Safety in Deep Reinforcement Learning using Unsupervised Action Planning","date":"2021-09-29","arxiv_id":"2109.14325","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-self-supervised-learning-for-vision","title":"Joint Self-Supervised Learning for Vision-based Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-attention-for-off-policy-actor-critic","title":"Meta Attention For Off-Policy Actor-Critic","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-batch-reinforcement-learning-via-sample","title":"Multi-batch Reinforcement Learning via Sample Transfer and Imitation Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-shifting-for-optimistic-exploration","title":"Reward Shifting for Optimistic Exploration and Conservative Exploitation","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"splid-self-imitation-policy-learning-through","title":"SPLID: Self-Imitation Policy Learning through Iterative Distillation","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"state-only-imitation-learning-by-trajectory","title":"State-Only Imitation Learning by Trajectory Distribution Matching","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-are-meta-reinforcement-learners","title":"Transformers are Meta-Reinforcement Learners","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"why-so-pessimistic-estimating-uncertainties","title":"Why so pessimistic? Estimating uncertainties for offline RL through ensembles, and why their independence matters.","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-more-when-it-needs-in-deep","title":"Exploring More When It Needs in Deep Reinforcement Learning","date":"2021-09-28","arxiv_id":"2109.13477","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-soft-actor-critic-mixing-prioritized","title":"Improved Soft Actor-Critic: Mixing Prioritized Off-Policy Samples with On-Policy Experience","date":"2021-09-24","arxiv_id":"2109.11767","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-ensemble-model-based-reinforcement","title":"Federated Ensemble Model-based Reinforcement Learning in Edge Computing","date":"2021-09-12","arxiv_id":"2109.05549","repositories_listed":0,"syntology":null},{"url":null,"slug":"ader-adapting-between-exploration-and","title":"ADER:Adapting between Exploration and Robustness for Actor-Critic Methods","date":"2021-09-08","arxiv_id":"2109.03443","repositories_listed":0,"syntology":null},{"url":null,"slug":"where-did-you-learn-that-from-surprising","title":"Membership Inference Attacks Against Temporally Correlated Data in Deep Reinforcement Learning","date":"2021-09-08","arxiv_id":"2109.03975","repositories_listed":0,"syntology":null},{"url":null,"slug":"photonic-quantum-policy-learning-in-openai","title":"Photonic Quantum Policy Learning in OpenAI Gym","date":"2021-08-29","arxiv_id":"2108.12926","repositories_listed":0,"syntology":null},{"url":null,"slug":"hac-explore-accelerating-exploration-with","title":"HAC Explore: Accelerating Exploration with Hierarchical Reinforcement Learning","date":"2021-08-12","arxiv_id":"2108.05872","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-based-reinforcement-learning-for","title":"Value-Based Reinforcement Learning for Continuous Control Robotic Manipulation in Multi-Task Sparse Reward Settings","date":"2021-07-28","arxiv_id":"2107.13356","repositories_listed":0,"syntology":null},{"url":null,"slug":"cautious-actor-critic","title":"Cautious Actor-Critic","date":"2021-07-12","arxiv_id":"2107.05217","repositories_listed":0,"syntology":null},{"url":null,"slug":"coordinate-wise-control-variates-for-deep","title":"Coordinate-wise Control Variates for Deep Policy Gradients","date":"2021-07-11","arxiv_id":"2107.04987","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-by-predicting-observations","title":"Imitation by Predicting Observations","date":"2021-07-08","arxiv_id":"2107.03851","repositories_listed":0,"syntology":null},{"url":null,"slug":"sa-matd3-self-attention-based-multi-agent","title":"SA-MATD3:Self-attention-based multi-agent continuous control method in cooperative environments","date":"2021-07-01","arxiv_id":"2107.00284","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-control-with-deep-reinforcement-1","title":"Continuous Control with Deep Reinforcement Learning for Autonomous Vessels","date":"2021-06-27","arxiv_id":"2106.14130","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-the-rain-from-removal-to","title":"Controlling the Rain: From Removal to Rendering","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"strategically-timed-state-observation-attacks","title":"Strategically-timed State-Observation Attacks on Deep Reinforcement Learning Agents","date":"2021-06-18","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-and-optimisation-of-bellman-residual","title":"Analysis and Optimisation of Bellman Residual Errors with Neural Function Approximation","date":"2021-06-16","arxiv_id":"2106.08774","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-sample-complexity-and-metastability-of","title":"On the Sample Complexity and Metastability of Heavy-tailed Policy Search in Continuous Control","date":"2021-06-15","arxiv_id":"2106.08414","repositories_listed":0,"syntology":null},{"url":null,"slug":"keyframe-focused-visual-imitation-learning","title":"Keyframe-Focused Visual Imitation Learning","date":"2021-06-11","arxiv_id":"2106.06452","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-as-anti","title":"Offline Reinforcement Learning as Anti-Exploration","date":"2021-06-11","arxiv_id":"2106.06431","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-battery-operation-for-energy","title":"Data-driven battery operation for energy arbitrage using rainbow deep reinforcement learning","date":"2021-06-10","arxiv_id":"2106.06061","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-bellman-operators","title":"Bayesian Bellman Operators","date":"2021-06-09","arxiv_id":"2106.05012","repositories_listed":0,"syntology":null},{"url":null,"slug":"average-reward-reinforcement-learning-with-2","title":"Average-Reward Reinforcement Learning with Trust Region Methods","date":"2021-06-07","arxiv_id":"2106.03442","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-prototype-of-reconfigurable-intelligent","title":"A Prototype of Reconfigurable Intelligent Surface with Continuous Control of the Reflection Phase","date":"2021-05-25","arxiv_id":"2105.11862","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperparameter-selection-for-imitation","title":"Hyperparameter Selection for Imitation Learning","date":"2021-05-25","arxiv_id":"2105.12034","repositories_listed":0,"syntology":null},{"url":null,"slug":"utilizing-skipped-frames-in-action-repeats","title":"Utilizing Skipped Frames in Action Repeats via Pseudo-Actions","date":"2021-05-07","arxiv_id":"2105.03041","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-dynamics-models-for-offline-1","title":"Autoregressive Dynamics Models for Offline Policy Evaluation and Optimization","date":"2021-04-28","arxiv_id":"2104.13877","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-manifold-search-exploring-the-manifold","title":"Policy Manifold Search: Exploring the Manifold Hypothesis for Diversity-based Neuroevolution","date":"2021-04-27","arxiv_id":"2104.13424","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-function-shape-exploration-in","title":"Reward function shape exploration in adversarial imitation learning: an empirical study","date":"2021-04-14","arxiv_id":"2104.06687","repositories_listed":0,"syntology":null},{"url":null,"slug":"gem-group-enhanced-model-for-learning","title":"GEM: Group Enhanced Model for Learning Dynamical Control Systems","date":"2021-04-07","arxiv_id":"2104.02844","repositories_listed":0,"syntology":null},{"url":null,"slug":"lazydagger-reducing-context-switching-in","title":"LazyDAgger: Reducing Context Switching in Interactive Imitation Learning","date":"2021-03-31","arxiv_id":"2104.00053","repositories_listed":0,"syntology":null},{"url":null,"slug":"hamiltonian-policy-optimization-in","title":"Hamiltonian Policy Optimization in Reinforcement Learning","date":"2021-03-23","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-actor-critic-reinforcement-learning","title":"Improving Actor-Critic Reinforcement Learning via Hamiltonian Monte Carlo Method","date":"2021-03-22","arxiv_id":"2103.12020","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-entropy-reinforcement-learning-with","title":"Maximum Entropy Reinforcement Learning with Mixture Policies","date":"2021-03-18","arxiv_id":"2103.10176","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-state-representations-via-temporal","title":"Learning State Representations via Temporal Cycle-Consistency Constraint in Model-Based Reinforcement Learning","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-based-mobile-robotics-obstacle","title":"Vision-Based Mobile Robotics Obstacle Avoidance With Deep Reinforcement Learning","date":"2021-03-08","arxiv_id":"2103.04727","repositories_listed":0,"syntology":null},{"url":null,"slug":"hamiltonian-policy-optimization","title":"Hamiltonian Policy Optimization","date":"2021-02-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-peng-s-q-l-for-modern","title":"Revisiting Peng's Q($λ$) for Modern Reinforcement Learning","date":"2021-02-27","arxiv_id":"2103.00107","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-precision-reinforcement-learning","title":"Low-Precision Reinforcement Learning: Running Soft Actor-Critic in Half Precision","date":"2021-02-26","arxiv_id":"2102.13565","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-of-twin-delayed-deep","title":"Twin actor twin delayed deep deterministic policy gradient (TATD3) learning for batch process control","date":"2021-02-25","arxiv_id":"2102.13012","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepthermal-combustion-optimization-for","title":"DeepThermal: Combustion Optimization for Thermal Power Generating Units Using Offline Reinforcement Learning","date":"2021-02-23","arxiv_id":"2102.11492","repositories_listed":0,"syntology":null},{"url":null,"slug":"gelato-geometrically-enriched-latent-model","title":"Uncertainty Estimation Using Riemannian Model~Dynamics for Offline Reinforcement Learning","date":"2021-02-22","arxiv_id":"2102.11327","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-proximal-policy-optimization-s-heavy-1","title":"On Proximal Policy Optimization's Heavy-tailed Gradients","date":"2021-02-20","arxiv_id":"2102.10264","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-invariant-state-abstractions-for-model","title":"Model-Invariant State Abstractions for Model-Based Reinforcement Learning","date":"2021-02-19","arxiv_id":"2102.09850","repositories_listed":0,"syntology":null},{"url":null,"slug":"closing-the-closed-loop-distribution-shift-in","title":"On the Sample Complexity of Stability Constrained Imitation Learning","date":"2021-02-18","arxiv_id":"2102.09161","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-memory-dependent-continuous-control","title":"Learning Memory-Dependent Continuous Control from Demonstrations","date":"2021-02-18","arxiv_id":"2102.09208","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-the-interaction-between-agents-in","title":"Modeling the Interaction between Agents in Cooperative Multi-Agent Reinforcement Learning","date":"2021-02-10","arxiv_id":"2102.06042","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-progress-in-deep-reinforcement-1","title":"Measuring Progress in Deep Reinforcement Learning Sample Efficiency","date":"2021-02-09","arxiv_id":"2102.04881","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-exploration-for-sample-efficient","title":"Decoupled Exploration and Exploitation Policies for Sample-Efficient Reinforcement Learning","date":"2021-01-23","arxiv_id":"2101.09458","repositories_listed":0,"syntology":null},{"url":null,"slug":"linear-representation-meta-reinforcement-1","title":"Linear Representation Meta-Reinforcement Learning for Instant Adaptation","date":"2021-01-12","arxiv_id":"2101.04750","repositories_listed":0,"syntology":null},{"url":null,"slug":"coachnet-an-adversarial-sampling-approach-for","title":"CoachNet: An Adversarial Sampling Approach for Reinforcement Learning","date":"2021-01-07","arxiv_id":"2101.02649","repositories_listed":0,"syntology":null},{"url":null,"slug":"derivative-free-policy-optimization-for-risk","title":"Derivative-Free Policy Optimization for Linear Risk-Sensitive and Robust Control Design: Implicit Regularization and Sample Complexity","date":"2021-01-04","arxiv_id":"2101.01041","repositories_listed":0,"syntology":null},{"url":null,"slug":"markov-chain-monte-carlo-policy-optimization","title":"Markov Chain Monte Carlo Policy Optimization","date":"2021-01-04","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-coherent-exploration-for-continuous","title":"Deep Coherent Exploration For Continuous Control","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-adaptive","title":"Deep Reinforcement Learning With Adaptive Combined Critics","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"divide-and-conquer-monte-carlo-tree-search-1","title":"Divide-and-Conquer Monte Carlo Tree Search","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"error-controlled-actor-critic-method-to","title":"Error Controlled Actor-Critic Method to Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"explicit-pareto-front-optimization-for","title":"Explicit Pareto Front Optimization for Constrained Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"factored-action-spaces-in-deep-reinforcement","title":"Factored Action Spaces in Deep Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"80a960a8382401dbf6e19df9a6047340c1f64db220eb197c70c820e52b9af91d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}