{"url":"/method/soft-actor-critic-autotuned-temperature","slug":"soft-actor-critic-autotuned-temperature","name":"Soft Actor-Critic (Autotuned Temperature)","full_name":"Soft Actor-Critic (Autotuned Temperature)","full_name_withheld":false,"description_markdown":"**Soft Actor Critic (Autotuned Temperature** is a modification of the [SAC](https://paperswithcode.com/method/soft-actor-critic) reinforcement learning algorithm. [SAC](https://paperswithcode.com/method/sac) can suffer from brittleness to the temperature hyperparameter. Unlike in conventional reinforcement learning, where the optimal policy is independent of scaling of the reward function, in maximum entropy reinforcement learning the scaling factor has to be compensated by the choice a of suitable temperature, and a sub-optimal temperature can drastically degrade performance. To resolve this issue, SAC with Autotuned Temperature has an automatic gradient-based temperature tuning method that adjusts the expected entropy over the visited states to match a target value.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Soft Actor-Critic Algorithms and Applications","paper":"/paper/soft-actor-critic-algorithms-and-applications","first_author":"Tuomas Haarnoja","n_authors":11,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/soft-actor-critic-algorithms-and-applications"},"source":{"url":"http://arxiv.org/abs/1812.05905v2","title":"Soft Actor-Critic Algorithms and Applications","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Reinforcement Learning","area_id":"reinforcement-learning","collection":"Policy Gradient Methods","url":"/methods/category/policy-gradient-methods","pwc_aliases":[]}],"n_papers_tagged":6,"archive_num_papers":6,"papers_newest_first":[{"paper":"/paper/leveraging-demonstrations-with-latent-space","title":"Leveraging Demonstrations with Latent Space Priors","date":"2022-10-26","arxiv_id":"2210.14685","n_code_links":1,"syntology":{"ran":2,"of":3,"unverified":1,"pointer_only":3}},{"paper":"/paper/soft-actor-critic-deep-reinforcement-learning","title":"Soft Actor-Critic Deep Reinforcement Learning for Fault Tolerant Flight Control","date":"2022-02-16","arxiv_id":"2202.09262","n_code_links":1,"syntology":null},{"paper":"/paper/self-supervised-policy-adaptation-during","title":"Self-Supervised Policy Adaptation during Deployment","date":"2020-07-08","arxiv_id":"2007.04309","n_code_links":2,"syntology":{"ran":16,"of":21,"unverified":5,"pointer_only":21}},{"paper":"/paper/particle-based-adaptive-discretization-for","title":"PFPN: Continuous Control of Physically Simulated Characters using Particle Filtering Policy Network","date":"2020-03-16","arxiv_id":"2003.06959","n_code_links":1,"syntology":null},{"paper":"/paper/discrete-and-continuous-action-representation","title":"Discrete and Continuous Action Representation for Practical RL in Video Games","date":"2019-12-23","arxiv_id":"1912.11077","n_code_links":1,"syntology":{"ran":3,"of":3,"unverified":0,"pointer_only":0}},{"paper":"/paper/soft-actor-critic-algorithms-and-applications","title":"Soft Actor-Critic Algorithms and Applications","date":"2018-12-13","arxiv_id":"1812.05905","n_code_links":52,"syntology":{"ran":10,"of":40,"unverified":30,"pointer_only":1}}],"papers_shown":6,"tasks":[{"task":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)","papers":4},{"task":"/task/reinforcement-learning","name":"Reinforcement Learning","papers":3},{"task":"/task/reinforcement-learning-2","name":"reinforcement-learning","papers":3},{"task":"/task/continuous-control","name":"Continuous Control","papers":1},{"task":"/task/control-with-prametrised-actions","name":"Control with Prametrised Actions","papers":1},{"task":"/task/decision-making","name":"Decision Making","papers":1},{"task":"/task/deep-reinforcement-learning","name":"Deep Reinforcement Learning","papers":1},{"task":"/task/offline-rl","name":"Offline RL","papers":1},{"task":"/task/continuous-control","name":"continuous-control","papers":1}],"tasks_shown":9,"n_tasks":9,"usage_by_year":[{"year":"2018","papers":1},{"year":"2019","papers":1},{"year":"2020","papers":2},{"year":"2022","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/soft-actor-critic-autotuned-temperature"},"syntology_read_at":"2026-09-25T09:33:49+00:00"}