{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/soft-actor-critic-off-policy-maximum-entropy","title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","arxiv_id":"1801.01290","date":"2018-01-04","proceeding":"ICML 2018 7","authors":["Tuomas Haarnoja","Aurick Zhou","Pieter Abbeel","Sergey Levine"],"abstract":"A platform for Applied Reinforcement Learning (Applied RL)","url_abs":"http://arxiv.org/abs/1801.01290v2","url_pdf":"http://arxiv.org/pdf/1801.01290v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/haarnoja/sac","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"NOASSERTION"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/AmmarFayad/Behavioral-Actor-Critic","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/AutumnWu/Streamlined-Off-Policy-Learning","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/FOCAL-ICLR/FOCAL-ICLR","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/Kaixhin/imitation-learning","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/MarsEleven/car_racer_RL","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/MrSyee/pg-is-all-you-need","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/QuentinVacher-rl/SoftActorCritic-in-Cpp-using-LibTorch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/RLAgent/state-marginal-matching","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/Rafael1s/Deep-Reinforcement-Learning-Udacity","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/SaminYeasar/off_policy_ac","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/ShawK91/Evolutionary-Reinforcement-Learning","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/ShawK91/erl_paper_nips18","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/Steinheilig/Imbiss","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"unanswered"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/X3N4/car_racer","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/ac-93/soft-actor-critic","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/ajaysub110/rl-pytorch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/andrejorsula/drl_grasping","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"BSD-3-Clause"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/araffin/sbx","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"jax","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/autumnwu/aggressive-q-learning-with-ensembles","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/baturaysaglam/la3p","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/ccolas/rl_stats","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/cindycia/Atari-SAC-Discrete","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/core-robotics-lab/icct","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/dasgringuen/assetto_corsa_gym","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/dfki-ric-underactuated-lab/torque_limited_simple_pendulum","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/donamin/llc","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"unanswered"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/facebookresearch/ReAgent","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/fdcl-gwu/gym-rotor","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/flowersteam/rl_stats","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/garyzyr001/rethinking-airl","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/gijskoning/ReproducingCURL","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/h-aboutalebi/SparceReward","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/hyunin-lee/ForecasterSAC","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/ikostrikov/jax-rl","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"jax","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/kairproject/kair_algorithms_draft","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/kdally/fault-tolerant-flight-control-drl","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/ku2482/discor.pytorch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/ku2482/gail-airl-ppo.pytorch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/ku2482/rljax","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"jax","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/ku2482/rltorch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/ku2482/soft-actor-critic.pytorch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/kushagra06/SAC","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/lanqingli1993/focal-iclr","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/learn-to-race/l2r","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"GPL-2.0"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/lollcat/Soft-Actor-Critic","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/lucadellalib/sac-beta","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/marload/DeepRL-TensorFlow2","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/moreanp/csro","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/mxblr/DeepRLHockey","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"unanswered"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/nagisazj/idaq_public","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/polixir/NeoRL","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"unanswered"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/pranz24/pytorch-soft-actor-critic","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/quantumiracle/Popular-RL-Algorithms","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/rk1998/robot-sac","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"unanswered"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/roythuly/obac","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/sunfex/weighted-sac","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/susan-amin/SparseBaseline1","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/tarod13/SAC","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/thomashirtz/pytorch-soft-actor-critic","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/thomashirtz/soft-actor-critic","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/tilkb/thermoai","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/timoklein/car_racer","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/tliu1997/rnac","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/tmjeong1103/RL_with_RAY","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/toshikwa/discor.pytorch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/toshikwa/soft-actor-critic.pytorch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/trackmania-rl/tmrl","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/watchernyu/spinningup-drl-prototyping","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/yhisaki/average-reward-drl","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/yimingpeng/sac-master","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"unanswered"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/yining043/SAC-discrete","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"unanswered"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/BY571/Soft-Actor-Critic-and-Extensions","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/DLR-RM/stable-baselines3","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/Ipsedo/EvoMotion","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/facebookresearch/rl/blob/main/examples/sac/sac.py","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"jax","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/hill-a/stable-baselines","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"tf","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/jakegrigsby/deep_control/blob/master/deep_control/sac.py","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/opendilab/DI-engine/blob/main/ding/policy/sac.py","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/pytorch/rl/tree/main/examples/sac","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"jax","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/ray-project/ray/tree/master/rllib","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"none","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/seungju-k1m/sac-td3-td7","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/tensorlayer/RLzoo","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"tf","reach":null},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/toni-sm/skrl","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"jax","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"soft-actor-critic-off-policy-maximum-entropy","repo_url":"https://github.com/xiuyu0000/new_papers_codes/tree/main/sac","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"mindspore","reach":null}],"tasks":[{"task_slug":"continuous-control","task_name":"Continuous Control"},{"task_slug":"decision-making","task_name":"Decision Making"},{"task_slug":"deep-reinforcement-learning","task_name":"Deep Reinforcement Learning"},{"task_slug":"omniverse-isaac-gym","task_name":"Omniverse Isaac Gym"},{"task_slug":"openai-gym","task_name":"OpenAI Gym"},{"task_slug":"q-learning","task_name":"Q-Learning"},{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[{"method_slug":"adam","method_name":"Adam"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"experience-replay","method_name":"Experience Replay"},{"method_slug":"relu","method_name":"ReLU"},{"method_slug":"soft-actor-critic","method_name":"Soft Actor Critic"}],"datasets_introduced":[],"methods_introduced":[{"slug":"soft-actor-critic","name":"Soft Actor Critic","full_name":"Soft Actor Critic"}],"results":[{"leaderboard":"/sota/continuous-control-on-lunar-lander-openai-gym","task":"Continuous Control","dataset":"Lunar Lander (OpenAI Gym)","model":"SAC","rank_in_archive_order":1,"of":5,"metrics":{"Score":"284.59±0.97"},"uses_additional_data":false},{"leaderboard":"/sota/openai-gym-on-ant-v4","task":"OpenAI Gym","dataset":"Ant-v4","model":"SAC","rank_in_archive_order":3,"of":5,"metrics":{"Average Return":"5208.09"},"uses_additional_data":false},{"leaderboard":"/sota/openai-gym-on-halfcheetah-v4","task":"OpenAI Gym","dataset":"HalfCheetah-v4","model":"SAC","rank_in_archive_order":1,"of":5,"metrics":{"Average Return":"15836.04"},"uses_additional_data":false},{"leaderboard":"/sota/openai-gym-on-hopper-v4","task":"OpenAI Gym","dataset":"Hopper-v4","model":"SAC","rank_in_archive_order":3,"of":5,"metrics":{"Average Return":"2882.56"},"uses_additional_data":false},{"leaderboard":"/sota/openai-gym-on-humanoid-v4","task":"OpenAI Gym","dataset":"Humanoid-v4","model":"SAC","rank_in_archive_order":2,"of":5,"metrics":{"Average Return":"6211.50"},"uses_additional_data":false},{"leaderboard":"/sota/openai-gym-on-walker2d-v4","task":"OpenAI Gym","dataset":"Walker2d-v4","model":"SAC","rank_in_archive_order":1,"of":5,"metrics":{"Average Return":"5745.27"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1801.01290","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1801.01290"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/hill-a/stable-baselines","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/haarnoja/sac","reach":{"status":"ok","spdx":"NOASSERTION"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/ku2482/soft-actor-critic.pytorch","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/DLR-RM/stable-baselines3","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/baturaysaglam/la3p","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/quantumiracle/Popular-RL-Algorithms","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/SaminYeasar/off_policy_ac","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/ShawK91/erl_paper_nips18","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/kairproject/kair_algorithms_draft","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/ku2482/rljax","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/ray-project/ray/tree/master/rllib","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/toni-sm/skrl","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Kaixhin/imitation-learning","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/dasgringuen/assetto_corsa_gym","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/watchernyu/spinningup-drl-prototyping","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/kushagra06/SAC","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/timoklein/car_racer","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/MrSyee/pg-is-all-you-need","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Rafael1s/Deep-Reinforcement-Learning-Udacity","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/polixir/NeoRL","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/core-robotics-lab/icct","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/tmjeong1103/RL_with_RAY","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/flowersteam/rl_stats","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/thomashirtz/pytorch-soft-actor-critic","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning","reach":null}],"summary":{"ran":69,"ran_draft_wrong":17,"ran_honours":3,"ran_fixture":1,"ran_violates":1,"unverified":57},"by_repo_kind":{"listed":{"samples":148,"ran":91,"repositories":27}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":66,"samples":[{"code_sha256_prefix":"424bf06f8d8eb998","entry":"AbstractAgent","repo":"h-aboutalebi/SparceReward","repo_kind":"listed","path":"engine/algorithms/SAC/sac.py","file_url":"https://github.com/h-aboutalebi/SparceReward/blob/HEAD/engine/algorithms/SAC/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"424bf06f8d8eb998"}},{"code_sha256_prefix":"de5e64dc2bfc6a33","entry":"Actor","repo":"MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning","repo_kind":"listed","path":"src/agents.py","file_url":"https://github.com/MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning/blob/HEAD/src/agents.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"de5e64dc2bfc6a33"}},{"code_sha256_prefix":"5d8f81c210a0aacb","entry":"Actor","repo":"BY571/Soft-Actor-Critic-and-Extensions","repo_kind":"listed","path":"files/Agent.py","file_url":"https://github.com/BY571/Soft-Actor-Critic-and-Extensions/blob/HEAD/files/Agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"5d8f81c210a0aacb"}},{"code_sha256_prefix":"dc01224a73ea8148","entry":"Actor","repo":"SaminYeasar/off_policy_ac","repo_kind":"listed","path":"SAC/SAC.py","file_url":"https://github.com/SaminYeasar/off_policy_ac/blob/HEAD/SAC/SAC.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"dc01224a73ea8148"}},{"code_sha256_prefix":"0918147b61a18074","entry":"Agent_ManualTemperature","repo":"MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning","repo_kind":"listed","path":"src/agents.py","file_url":"https://github.com/MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning/blob/HEAD/src/agents.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"0918147b61a18074"}},{"code_sha256_prefix":"d6b54769dcaa3ec5","entry":"BaseQNet","repo":"sunfex/weighted-sac","repo_kind":"listed","path":"sac.py","file_url":"https://github.com/sunfex/weighted-sac/blob/HEAD/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"d6b54769dcaa3ec5"}},{"code_sha256_prefix":"6bb425a4771dc88f","entry":"Batch","repo":"yhisaki/average-reward-drl","repo_kind":"listed","path":"average_reward_drl/algorithms/sac.py","file_url":"https://github.com/yhisaki/average-reward-drl/blob/HEAD/average_reward_drl/algorithms/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"6bb425a4771dc88f"}},{"code_sha256_prefix":"e6f3bef6186ec524","entry":"BatchQNet","repo":"sunfex/weighted-sac","repo_kind":"listed","path":"sac.py","file_url":"https://github.com/sunfex/weighted-sac/blob/HEAD/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"e6f3bef6186ec524"}},{"code_sha256_prefix":"94573a894a42ebff","entry":"Buffer","repo":"garyzyr001/rethinking-airl","repo_kind":"listed","path":"airl/algo/sac.py","file_url":"https://github.com/garyzyr001/rethinking-airl/blob/HEAD/airl/algo/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"94573a894a42ebff"}},{"code_sha256_prefix":"c10734f04f231cb4","entry":"ConcatStateAction","repo":"yhisaki/average-reward-drl","repo_kind":"listed","path":"average_reward_drl/algorithms/sac.py","file_url":"https://github.com/yhisaki/average-reward-drl/blob/HEAD/average_reward_drl/algorithms/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"c10734f04f231cb4"}},{"code_sha256_prefix":"7c4433b21cf61a2c","entry":"Critic","repo":"MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning","repo_kind":"listed","path":"src/agents.py","file_url":"https://github.com/MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning/blob/HEAD/src/agents.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"7c4433b21cf61a2c"}},{"code_sha256_prefix":"f8ead4c2c205ecd8","entry":"Critic","repo":"baturaysaglam/la3p","repo_kind":"listed","path":"Code/SAC/SAC.py","file_url":"https://github.com/baturaysaglam/la3p/blob/HEAD/Code/SAC/SAC.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"f8ead4c2c205ecd8"}},{"code_sha256_prefix":"8f0377d2cad944f3","entry":"Critic","repo":"BY571/Soft-Actor-Critic-and-Extensions","repo_kind":"listed","path":"files/Agent.py","file_url":"https://github.com/BY571/Soft-Actor-Critic-and-Extensions/blob/HEAD/files/Agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"8f0377d2cad944f3"}},{"code_sha256_prefix":"a0a4c6c892e497cd","entry":"Critic","repo":"SaminYeasar/off_policy_ac","repo_kind":"listed","path":"SAC/SAC.py","file_url":"https://github.com/SaminYeasar/off_policy_ac/blob/HEAD/SAC/SAC.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"a0a4c6c892e497cd"}},{"code_sha256_prefix":"9ee95d5b4fc2dd44","entry":"DeepActor","repo":"BY571/Soft-Actor-Critic-and-Extensions","repo_kind":"listed","path":"files/Agent.py","file_url":"https://github.com/BY571/Soft-Actor-Critic-and-Extensions/blob/HEAD/files/Agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"9ee95d5b4fc2dd44"}},{"code_sha256_prefix":"a57bb45ab6a0bc51","entry":"DeepCritic","repo":"BY571/Soft-Actor-Critic-and-Extensions","repo_kind":"listed","path":"files/Agent.py","file_url":"https://github.com/BY571/Soft-Actor-Critic-and-Extensions/blob/HEAD/files/Agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"a57bb45ab6a0bc51"}},{"code_sha256_prefix":"693c33a1aa6474c6","entry":"DeepIQN","repo":"BY571/Soft-Actor-Critic-and-Extensions","repo_kind":"listed","path":"files/Agent.py","file_url":"https://github.com/BY571/Soft-Actor-Critic-and-Extensions/blob/HEAD/files/Agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"693c33a1aa6474c6"}},{"code_sha256_prefix":"7553f61c31debefb","entry":"DeterministicPolicy","repo":"baturaysaglam/la3p","repo_kind":"listed","path":"Code/SAC/SAC.py","file_url":"https://github.com/baturaysaglam/la3p/blob/HEAD/Code/SAC/SAC.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"7553f61c31debefb"}},{"code_sha256_prefix":"3faeea6e22edd34d","entry":"DeterministicPolicy","repo":"h-aboutalebi/SparceReward","repo_kind":"listed","path":"engine/algorithms/SAC/sac.py","file_url":"https://github.com/h-aboutalebi/SparceReward/blob/HEAD/engine/algorithms/SAC/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"3faeea6e22edd34d"}},{"code_sha256_prefix":"d4dfbbf39952f704","entry":"DeterministicPolicy","repo":"MarsEleven/car_racer_RL","repo_kind":"listed","path":"sac/sac.py","file_url":"https://github.com/MarsEleven/car_racer_RL/blob/HEAD/sac/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"d4dfbbf39952f704"}},{"code_sha256_prefix":"bf0d89394b22e1b6","entry":"GaussianPolicy","repo":"Rafael1s/Deep-Reinforcement-Learning-Udacity","repo_kind":"listed","path":"Ant-PyBulletEnv-Soft-Actor-Critic/sac_agent.py","file_url":"https://github.com/Rafael1s/Deep-Reinforcement-Learning-Udacity/blob/HEAD/Ant-PyBulletEnv-Soft-Actor-Critic/sac_agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"bf0d89394b22e1b6"}},{"code_sha256_prefix":"152bc1e449547d23","entry":"GaussianPolicy","repo":"baturaysaglam/la3p","repo_kind":"listed","path":"Code/SAC/SAC.py","file_url":"https://github.com/baturaysaglam/la3p/blob/HEAD/Code/SAC/SAC.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"152bc1e449547d23"}},{"code_sha256_prefix":"8761dc745b44dc04","entry":"GaussianPolicy","repo":"ku2482/discor.pytorch","repo_kind":"listed","path":"discor/algorithm/sac.py","file_url":"https://github.com/ku2482/discor.pytorch/blob/HEAD/discor/algorithm/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"8761dc745b44dc04"}},{"code_sha256_prefix":"8ee15fddb3528e86","entry":"GaussianPolicy","repo":"h-aboutalebi/SparceReward","repo_kind":"listed","path":"engine/algorithms/SAC/sac.py","file_url":"https://github.com/h-aboutalebi/SparceReward/blob/HEAD/engine/algorithms/SAC/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"8ee15fddb3528e86"}},{"code_sha256_prefix":"da8ebd6e9ea11fca","entry":"GaussianPolicy","repo":"roythuly/obac","repo_kind":"listed","path":"model/algo.py","file_url":"https://github.com/roythuly/obac/blob/HEAD/model/algo.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"da8ebd6e9ea11fca"}},{"code_sha256_prefix":"5722a85a4a153169","entry":"GaussianPolicy","repo":"sunfex/weighted-sac","repo_kind":"listed","path":"sac.py","file_url":"https://github.com/sunfex/weighted-sac/blob/HEAD/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"5722a85a4a153169"}},{"code_sha256_prefix":"57c7562fda9bbab3","entry":"GaussianPolicy","repo":"MarsEleven/car_racer_RL","repo_kind":"listed","path":"sac/sac.py","file_url":"https://github.com/MarsEleven/car_racer_RL/blob/HEAD/sac/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"57c7562fda9bbab3"}},{"code_sha256_prefix":"94a135edcb95baf2","entry":"IQN","repo":"BY571/Soft-Actor-Critic-and-Extensions","repo_kind":"listed","path":"files/Agent.py","file_url":"https://github.com/BY571/Soft-Actor-Critic-and-Extensions/blob/HEAD/files/Agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"94a135edcb95baf2"}},{"code_sha256_prefix":"4aec8353051fa16d","entry":"Logger","repo":"yhisaki/average-reward-drl","repo_kind":"listed","path":"average_reward_drl/algorithms/sac.py","file_url":"https://github.com/yhisaki/average-reward-drl/blob/HEAD/average_reward_drl/algorithms/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"4aec8353051fa16d"}},{"code_sha256_prefix":"a830c8ece63f9c57","entry":"MLPActorCritic","repo":"tmjeong1103/RL_with_RAY","repo_kind":"listed","path":"SAC_RAY/model.py","file_url":"https://github.com/tmjeong1103/RL_with_RAY/blob/HEAD/SAC_RAY/model.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"a830c8ece63f9c57"}},{"code_sha256_prefix":"c02fc6e11c3ba5cf","entry":"MLPGaussianPolicy","repo":"tmjeong1103/RL_with_RAY","repo_kind":"listed","path":"SAC_RAY/model.py","file_url":"https://github.com/tmjeong1103/RL_with_RAY/blob/HEAD/SAC_RAY/model.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"c02fc6e11c3ba5cf"}},{"code_sha256_prefix":"14211bf5685457cf","entry":"MLPQFunction","repo":"tmjeong1103/RL_with_RAY","repo_kind":"listed","path":"SAC_RAY/model.py","file_url":"https://github.com/tmjeong1103/RL_with_RAY/blob/HEAD/SAC_RAY/model.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"14211bf5685457cf"}},{"code_sha256_prefix":"5d3a98fbf8446e0c","entry":"MultiLinear","repo":"yhisaki/average-reward-drl","repo_kind":"listed","path":"average_reward_drl/algorithms/sac.py","file_url":"https://github.com/yhisaki/average-reward-drl/blob/HEAD/average_reward_drl/algorithms/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"5d3a98fbf8446e0c"}},{"code_sha256_prefix":"de764e9177884a12","entry":"Network","repo":"MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning","repo_kind":"listed","path":"src/agents.py","file_url":"https://github.com/MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning/blob/HEAD/src/agents.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"de764e9177884a12"}},{"code_sha256_prefix":"540316788d2c30c7","entry":"PolicyNetwork","repo":"quantumiracle/Popular-RL-Algorithms","repo_kind":"listed","path":"sac_v2.py","file_url":"https://github.com/quantumiracle/Popular-RL-Algorithms/blob/HEAD/sac_v2.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"540316788d2c30c7"}},{"code_sha256_prefix":"ec7050db6c32a43a","entry":"PolicyNetwork","repo":"ajaysub110/rl-pytorch","repo_kind":"listed","path":"sac/models/soft_actor_critic.py","file_url":"https://github.com/ajaysub110/rl-pytorch/blob/HEAD/sac/models/soft_actor_critic.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"ec7050db6c32a43a"}},{"code_sha256_prefix":"91bcdaabd64f8eee","entry":"PrioritizedReplay","repo":"BY571/Soft-Actor-Critic-and-Extensions","repo_kind":"listed","path":"files/Agent.py","file_url":"https://github.com/BY571/Soft-Actor-Critic-and-Extensions/blob/HEAD/files/Agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"91bcdaabd64f8eee"}},{"code_sha256_prefix":"f37748964b0cd7da","entry":"QNetwork","repo":"Rafael1s/Deep-Reinforcement-Learning-Udacity","repo_kind":"listed","path":"Ant-PyBulletEnv-Soft-Actor-Critic/sac_agent.py","file_url":"https://github.com/Rafael1s/Deep-Reinforcement-Learning-Udacity/blob/HEAD/Ant-PyBulletEnv-Soft-Actor-Critic/sac_agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"f37748964b0cd7da"}},{"code_sha256_prefix":"b7030d8d56cd639a","entry":"QNetwork","repo":"thomashirtz/soft-actor-critic","repo_kind":"listed","path":"soft_actor_critic/agent.py","file_url":"https://github.com/thomashirtz/soft-actor-critic/blob/HEAD/soft_actor_critic/agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"b7030d8d56cd639a"}},{"code_sha256_prefix":"0d80189fd77d3aa6","entry":"QNetwork","repo":"roythuly/obac","repo_kind":"listed","path":"model/algo.py","file_url":"https://github.com/roythuly/obac/blob/HEAD/model/algo.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"0d80189fd77d3aa6"}},{"code_sha256_prefix":"e3aca820c0049604","entry":"QNetwork","repo":"MarsEleven/car_racer_RL","repo_kind":"listed","path":"sac/sac.py","file_url":"https://github.com/MarsEleven/car_racer_RL/blob/HEAD/sac/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"e3aca820c0049604"}},{"code_sha256_prefix":"e182a5c8f1d1e16f","entry":"QNetwork","repo":"ajaysub110/rl-pytorch","repo_kind":"listed","path":"sac/models/soft_actor_critic.py","file_url":"https://github.com/ajaysub110/rl-pytorch/blob/HEAD/sac/models/soft_actor_critic.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"e182a5c8f1d1e16f"}},{"code_sha256_prefix":"ac5ecca5b7705f8a","entry":"ReplayBuffer","repo":"MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning","repo_kind":"listed","path":"src/agents.py","file_url":"https://github.com/MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning/blob/HEAD/src/agents.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"ac5ecca5b7705f8a"}},{"code_sha256_prefix":"8119b0afbe2d940e","entry":"ReplayBuffer","repo":"kdally/fault-tolerant-flight-control-drl","repo_kind":"listed","path":"fault_tolerant_flight_control_drl/agent/sac.py","file_url":"https://github.com/kdally/fault-tolerant-flight-control-drl/blob/HEAD/fault_tolerant_flight_control_drl/agent/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"8119b0afbe2d940e"}},{"code_sha256_prefix":"a6c39d1200737d3f","entry":"ReplayBuffer","repo":"BY571/Soft-Actor-Critic-and-Extensions","repo_kind":"listed","path":"files/Agent.py","file_url":"https://github.com/BY571/Soft-Actor-Critic-and-Extensions/blob/HEAD/files/Agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"a6c39d1200737d3f"}},{"code_sha256_prefix":"e9664791458e5ade","entry":"ReplayBuffer","repo":"yhisaki/average-reward-drl","repo_kind":"listed","path":"average_reward_drl/algorithms/sac.py","file_url":"https://github.com/yhisaki/average-reward-drl/blob/HEAD/average_reward_drl/algorithms/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"e9664791458e5ade"}},{"code_sha256_prefix":"70249df825c62d7e","entry":"ReplayBuffer","repo":"quantumiracle/Popular-RL-Algorithms","repo_kind":"listed","path":"sac_v2.py","file_url":"https://github.com/quantumiracle/Popular-RL-Algorithms/blob/HEAD/sac_v2.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"70249df825c62d7e"}},{"code_sha256_prefix":"0020d85f4dbfb10b","entry":"ReplayBuffer","repo":"thomashirtz/soft-actor-critic","repo_kind":"listed","path":"soft_actor_critic/agent.py","file_url":"https://github.com/thomashirtz/soft-actor-critic/blob/HEAD/soft_actor_critic/agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"0020d85f4dbfb10b"}},{"code_sha256_prefix":"217cd7f6f86864da","entry":"SAC","repo":"ShawK91/Evolutionary-Reinforcement-Learning","repo_kind":"listed","path":"algos/sac.py","file_url":"https://github.com/ShawK91/Evolutionary-Reinforcement-Learning/blob/HEAD/algos/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"217cd7f6f86864da"}},{"code_sha256_prefix":"43d63d9b3b6240a1","entry":"SAC","repo":"SaminYeasar/off_policy_ac","repo_kind":"listed","path":"SAC/SAC.py","file_url":"https://github.com/SaminYeasar/off_policy_ac/blob/HEAD/SAC/SAC.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"43d63d9b3b6240a1"}},{"code_sha256_prefix":"e5d43066162708ff","entry":"SACActor","repo":"kushagra06/SAC","repo_kind":"listed","path":"model.py","file_url":"https://github.com/kushagra06/SAC/blob/HEAD/model.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"e5d43066162708ff"}},{"code_sha256_prefix":"33ec92b652664841","entry":"ScalarHolder","repo":"yhisaki/average-reward-drl","repo_kind":"listed","path":"average_reward_drl/algorithms/sac.py","file_url":"https://github.com/yhisaki/average-reward-drl/blob/HEAD/average_reward_drl/algorithms/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"33ec92b652664841"}},{"code_sha256_prefix":"c74fba5d30037b69","entry":"SerializedBuffer","repo":"garyzyr001/rethinking-airl","repo_kind":"listed","path":"airl/algo/sac.py","file_url":"https://github.com/garyzyr001/rethinking-airl/blob/HEAD/airl/algo/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"c74fba5d30037b69"}},{"code_sha256_prefix":"eb9bc09b3ca641cb","entry":"SerializedBuffer","repo":"ku2482/gail-airl-ppo.pytorch","repo_kind":"listed","path":"gail_airl_ppo/algo/sac.py","file_url":"https://github.com/ku2482/gail-airl-ppo.pytorch/blob/HEAD/gail_airl_ppo/algo/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"eb9bc09b3ca641cb"}},{"code_sha256_prefix":"b473c449b3da39cc","entry":"SoftActor","repo":"Kaixhin/imitation-learning","repo_kind":"listed","path":"models.py","file_url":"https://github.com/Kaixhin/imitation-learning/blob/HEAD/models.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"b473c449b3da39cc"}},{"code_sha256_prefix":"0688d4da92feee38","entry":"SoftQNetwork","repo":"quantumiracle/Popular-RL-Algorithms","repo_kind":"listed","path":"sac_v2.py","file_url":"https://github.com/quantumiracle/Popular-RL-Algorithms/blob/HEAD/sac_v2.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"0688d4da92feee38"}},{"code_sha256_prefix":"02a45597a277afab","entry":"SquashedDiagonalGaussianHead","repo":"yhisaki/average-reward-drl","repo_kind":"listed","path":"average_reward_drl/algorithms/sac.py","file_url":"https://github.com/yhisaki/average-reward-drl/blob/HEAD/average_reward_drl/algorithms/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"02a45597a277afab"}},{"code_sha256_prefix":"b1e066bc1ff4b61b","entry":"StateActionFunction","repo":"ku2482/discor.pytorch","repo_kind":"listed","path":"discor/algorithm/sac.py","file_url":"https://github.com/ku2482/discor.pytorch/blob/HEAD/discor/algorithm/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"b1e066bc1ff4b61b"}},{"code_sha256_prefix":"f0f9d0b00b3e9632","entry":"StateDependentPolicy","repo":"garyzyr001/rethinking-airl","repo_kind":"listed","path":"airl/algo/sac.py","file_url":"https://github.com/garyzyr001/rethinking-airl/blob/HEAD/airl/algo/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"f0f9d0b00b3e9632"}},{"code_sha256_prefix":"77baa7a1a168fc24","entry":"StateDependentPolicy","repo":"ku2482/gail-airl-ppo.pytorch","repo_kind":"listed","path":"gail_airl_ppo/algo/sac.py","file_url":"https://github.com/ku2482/gail-airl-ppo.pytorch/blob/HEAD/gail_airl_ppo/algo/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"77baa7a1a168fc24"}},{"code_sha256_prefix":"7e61ac4f70c90628","entry":"StochasticPolicy","repo":"thomashirtz/soft-actor-critic","repo_kind":"listed","path":"soft_actor_critic/agent.py","file_url":"https://github.com/thomashirtz/soft-actor-critic/blob/HEAD/soft_actor_critic/agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"7e61ac4f70c90628"}},{"code_sha256_prefix":"f552f5cc13186908","entry":"TanhGaussianPolicySACAdapt","repo":"autumnwu/aggressive-q-learning-with-ensembles","repo_kind":"listed","path":"spinup/algos/aqe/core_aqe.py","file_url":"https://github.com/autumnwu/aggressive-q-learning-with-ensembles/blob/HEAD/spinup/algos/aqe/core_aqe.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"f552f5cc13186908"}},{"code_sha256_prefix":"5b2e9a4c85bbab7d","entry":"TwinnedQNetworks","repo":"thomashirtz/soft-actor-critic","repo_kind":"listed","path":"soft_actor_critic/agent.py","file_url":"https://github.com/thomashirtz/soft-actor-critic/blob/HEAD/soft_actor_critic/agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"5b2e9a4c85bbab7d"}},{"code_sha256_prefix":"5d13132d2e3c5ef0","entry":"TwinnedStateActionFunction","repo":"ku2482/discor.pytorch","repo_kind":"listed","path":"discor/algorithm/sac.py","file_url":"https://github.com/ku2482/discor.pytorch/blob/HEAD/discor/algorithm/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"5d13132d2e3c5ef0"}},{"code_sha256_prefix":"e42d91ba746c3a14","entry":"TwinnedStateActionFunction","repo":"garyzyr001/rethinking-airl","repo_kind":"listed","path":"airl/algo/sac.py","file_url":"https://github.com/garyzyr001/rethinking-airl/blob/HEAD/airl/algo/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"e42d91ba746c3a14"}},{"code_sha256_prefix":"f8845c35735f20bc","entry":"Value","repo":"MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning","repo_kind":"listed","path":"src/agents.py","file_url":"https://github.com/MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning/blob/HEAD/src/agents.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"f8845c35735f20bc"}},{"code_sha256_prefix":"18ecb1311366b8cb","entry":"ValueNetwork","repo":"roythuly/obac","repo_kind":"listed","path":"model/algo.py","file_url":"https://github.com/roythuly/obac/blob/HEAD/model/algo.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"18ecb1311366b8cb"}},{"code_sha256_prefix":"b7fc01f37cbb0eb1","entry":"_create_fcnn","repo":"Kaixhin/imitation-learning","repo_kind":"listed","path":"models.py","file_url":"https://github.com/Kaixhin/imitation-learning/blob/HEAD/models.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"b7fc01f37cbb0eb1"}},{"code_sha256_prefix":"52028903a218ba91","entry":"apply_squashing_func","repo":"SaminYeasar/off_policy_ac","repo_kind":"listed","path":"SAC/SAC.py","file_url":"https://github.com/SaminYeasar/off_policy_ac/blob/HEAD/SAC/SAC.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"52028903a218ba91"}},{"code_sha256_prefix":"fa8517ef05057c7f","entry":"bytes_to_params","repo":"kdally/fault-tolerant-flight-control-drl","repo_kind":"listed","path":"fault_tolerant_flight_control_drl/agent/sac.py","file_url":"https://github.com/kdally/fault-tolerant-flight-control-drl/blob/HEAD/fault_tolerant_flight_control_drl/agent/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"fa8517ef05057c7f"}},{"code_sha256_prefix":"d9d455622c2dcd61","entry":"calculate_huber_loss","repo":"BY571/Soft-Actor-Critic-and-Extensions","repo_kind":"listed","path":"files/Agent.py","file_url":"https://github.com/BY571/Soft-Actor-Critic-and-Extensions/blob/HEAD/files/Agent.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"d9d455622c2dcd61"}},{"code_sha256_prefix":"4731ca1d377b8284","entry":"clip_but_pass_gradient","repo":"watchernyu/spinningup-drl-prototyping","repo_kind":"listed","path":"spinup/algos/sac/core.py","file_url":"https://github.com/watchernyu/spinningup-drl-prototyping/blob/HEAD/spinup/algos/sac/core.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"4731ca1d377b8284"}},{"code_sha256_prefix":"a2697a9ea209c680","entry":"compute_weight_X","repo":"hyunin-lee/ForecasterSAC","repo_kind":"listed","path":"sac.py","file_url":"https://github.com/hyunin-lee/ForecasterSAC/blob/HEAD/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"a2697a9ea209c680"}},{"code_sha256_prefix":"cdf94e91c8746537","entry":"convert_to_tensor","repo":"yhisaki/average-reward-drl","repo_kind":"listed","path":"average_reward_drl/algorithms/sac.py","file_url":"https://github.com/yhisaki/average-reward-drl/blob/HEAD/average_reward_drl/algorithms/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"cdf94e91c8746537"}},{"code_sha256_prefix":"248f63c9cf4d972e","entry":"create_linear_network","repo":"ku2482/discor.pytorch","repo_kind":"listed","path":"discor/algorithm/sac.py","file_url":"https://github.com/ku2482/discor.pytorch/blob/HEAD/discor/algorithm/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"248f63c9cf4d972e"}},{"code_sha256_prefix":"36b66092679a3132","entry":"data_to_json","repo":"kdally/fault-tolerant-flight-control-drl","repo_kind":"listed","path":"fault_tolerant_flight_control_drl/agent/sac.py","file_url":"https://github.com/kdally/fault-tolerant-flight-control-drl/blob/HEAD/fault_tolerant_flight_control_drl/agent/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"none","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"36b66092679a3132"}},{"code_sha256_prefix":"298e84cf1d9ed36c","entry":"eval_mode","repo":"thomashirtz/soft-actor-critic","repo_kind":"listed","path":"soft_actor_critic/agent.py","file_url":"https://github.com/thomashirtz/soft-actor-critic/blob/HEAD/soft_actor_critic/agent.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"298e84cf1d9ed36c"}},{"code_sha256_prefix":"5c26c93931dce503","entry":"gaussian_likelihood","repo":"watchernyu/spinningup-drl-prototyping","repo_kind":"listed","path":"spinup/algos/sac/core.py","file_url":"https://github.com/watchernyu/spinningup-drl-prototyping/blob/HEAD/spinup/algos/sac/core.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"5c26c93931dce503"}},{"code_sha256_prefix":"958cab9f3ba6af81","entry":"gaussian_likelihood","repo":"SaminYeasar/off_policy_ac","repo_kind":"listed","path":"SAC/SAC.py","file_url":"https://github.com/SaminYeasar/off_policy_ac/blob/HEAD/SAC/SAC.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"958cab9f3ba6af81"}},{"code_sha256_prefix":"d4978950888874f4","entry":"get_device","repo":"thomashirtz/soft-actor-critic","repo_kind":"listed","path":"soft_actor_critic/agent.py","file_url":"https://github.com/thomashirtz/soft-actor-critic/blob/HEAD/soft_actor_critic/agent.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"d4978950888874f4"}},{"code_sha256_prefix":"1afbe631504fbaf7","entry":"get_multilayer_perceptron","repo":"thomashirtz/soft-actor-critic","repo_kind":"listed","path":"soft_actor_critic/agent.py","file_url":"https://github.com/thomashirtz/soft-actor-critic/blob/HEAD/soft_actor_critic/agent.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"1afbe631504fbaf7"}},{"code_sha256_prefix":"1dd9b9ae9fa9426f","entry":"group_by_keys","repo":"yhisaki/average-reward-drl","repo_kind":"listed","path":"average_reward_drl/algorithms/sac.py","file_url":"https://github.com/yhisaki/average-reward-drl/blob/HEAD/average_reward_drl/algorithms/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"1dd9b9ae9fa9426f"}},{"code_sha256_prefix":"c70b5621a0ca8047","entry":"is_json_serializable","repo":"kdally/fault-tolerant-flight-control-drl","repo_kind":"listed","path":"fault_tolerant_flight_control_drl/agent/sac.py","file_url":"https://github.com/kdally/fault-tolerant-flight-control-drl/blob/HEAD/fault_tolerant_flight_control_drl/agent/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"c70b5621a0ca8047"}},{"code_sha256_prefix":"ad86f46cc9af956b","entry":"json_to_data","repo":"kdally/fault-tolerant-flight-control-drl","repo_kind":"listed","path":"fault_tolerant_flight_control_drl/agent/sac.py","file_url":"https://github.com/kdally/fault-tolerant-flight-control-drl/blob/HEAD/fault_tolerant_flight_control_drl/agent/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"ad86f46cc9af956b"}},{"code_sha256_prefix":"a1e9dc4773ffeba5","entry":"mlp","repo":"tmjeong1103/RL_with_RAY","repo_kind":"listed","path":"SAC_RAY/model.py","file_url":"https://github.com/tmjeong1103/RL_with_RAY/blob/HEAD/SAC_RAY/model.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"a1e9dc4773ffeba5"}},{"code_sha256_prefix":"5c5bce3523ed4d69","entry":"ortho_init","repo":"yhisaki/average-reward-drl","repo_kind":"listed","path":"average_reward_drl/algorithms/sac.py","file_url":"https://github.com/yhisaki/average-reward-drl/blob/HEAD/average_reward_drl/algorithms/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"5c5bce3523ed4d69"}},{"code_sha256_prefix":"1d75daad61f15424","entry":"params_to_bytes","repo":"kdally/fault-tolerant-flight-control-drl","repo_kind":"listed","path":"fault_tolerant_flight_control_drl/agent/sac.py","file_url":"https://github.com/kdally/fault-tolerant-flight-control-drl/blob/HEAD/fault_tolerant_flight_control_drl/agent/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"1d75daad61f15424"}},{"code_sha256_prefix":"642892e9419a125d","entry":"policyNet","repo":"tarod13/SAC","repo_kind":"listed","path":"nets.py","file_url":"https://github.com/tarod13/SAC/blob/HEAD/nets.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"642892e9419a125d"}},{"code_sha256_prefix":"4a486e5048778576","entry":"scale_action","repo":"kdally/fault-tolerant-flight-control-drl","repo_kind":"listed","path":"fault_tolerant_flight_control_drl/agent/sac.py","file_url":"https://github.com/kdally/fault-tolerant-flight-control-drl/blob/HEAD/fault_tolerant_flight_control_drl/agent/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"4a486e5048778576"}},{"code_sha256_prefix":"3a0835e6ce008741","entry":"soft_actor_critic_agent","repo":"Rafael1s/Deep-Reinforcement-Learning-Udacity","repo_kind":"listed","path":"Ant-PyBulletEnv-Soft-Actor-Critic/sac_agent.py","file_url":"https://github.com/Rafael1s/Deep-Reinforcement-Learning-Udacity/blob/HEAD/Ant-PyBulletEnv-Soft-Actor-Critic/sac_agent.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"3a0835e6ce008741"}},{"code_sha256_prefix":"5a4d3fd6b28ecaf4","entry":"unscale_action","repo":"kdally/fault-tolerant-flight-control-drl","repo_kind":"listed","path":"fault_tolerant_flight_control_drl/agent/sac.py","file_url":"https://github.com/kdally/fault-tolerant-flight-control-drl/blob/HEAD/fault_tolerant_flight_control_drl/agent/sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"5a4d3fd6b28ecaf4"}},{"code_sha256_prefix":"b0e759b2d521e146","entry":"Agent","repo":"MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning","repo_kind":"listed","path":"src/agents.py","file_url":"https://github.com/MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning/blob/HEAD/src/agents.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"b0e759b2d521e146"}},{"code_sha256_prefix":"2ae117303c9ac594","entry":"Agent","repo":"BY571/Soft-Actor-Critic-and-Extensions","repo_kind":"listed","path":"files/Agent.py","file_url":"https://github.com/BY571/Soft-Actor-Critic-and-Extensions/blob/HEAD/files/Agent.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"2ae117303c9ac594"}},{"code_sha256_prefix":"05437b66f08eb4d6","entry":"Agent","repo":"thomashirtz/soft-actor-critic","repo_kind":"listed","path":"soft_actor_critic/agent.py","file_url":"https://github.com/thomashirtz/soft-actor-critic/blob/HEAD/soft_actor_critic/agent.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"05437b66f08eb4d6"}},{"code_sha256_prefix":"eaf547baf5f33e43","entry":"Algorithm","repo":"ku2482/discor.pytorch","repo_kind":"listed","path":"discor/algorithm/sac.py","file_url":"https://github.com/ku2482/discor.pytorch/blob/HEAD/discor/algorithm/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"eaf547baf5f33e43"}},{"code_sha256_prefix":"803d90cd603ff8c0","entry":"Algorithm","repo":"garyzyr001/rethinking-airl","repo_kind":"listed","path":"airl/algo/sac.py","file_url":"https://github.com/garyzyr001/rethinking-airl/blob/HEAD/airl/algo/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"803d90cd603ff8c0"}},{"code_sha256_prefix":"ff7a80131e64a08d","entry":"Algorithm","repo":"ku2482/gail-airl-ppo.pytorch","repo_kind":"listed","path":"gail_airl_ppo/algo/sac.py","file_url":"https://github.com/ku2482/gail-airl-ppo.pytorch/blob/HEAD/gail_airl_ppo/algo/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"ff7a80131e64a08d"}},{"code_sha256_prefix":"110041979f3c0cb8","entry":"AlgorithmBase","repo":"yhisaki/average-reward-drl","repo_kind":"listed","path":"average_reward_drl/algorithms/sac.py","file_url":"https://github.com/yhisaki/average-reward-drl/blob/HEAD/average_reward_drl/algorithms/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"110041979f3c0cb8"}},{"code_sha256_prefix":"d6d4b5a5d84ba8ad","entry":"BaseNetwork","repo":"ku2482/discor.pytorch","repo_kind":"listed","path":"discor/algorithm/sac.py","file_url":"https://github.com/ku2482/discor.pytorch/blob/HEAD/discor/algorithm/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"d6d4b5a5d84ba8ad"}},{"code_sha256_prefix":"b6b1a23b07a627dd","entry":"Buffer","repo":"ku2482/gail-airl-ppo.pytorch","repo_kind":"listed","path":"gail_airl_ppo/algo/sac.py","file_url":"https://github.com/ku2482/gail-airl-ppo.pytorch/blob/HEAD/gail_airl_ppo/algo/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"b6b1a23b07a627dd"}},{"code_sha256_prefix":"3ef56075bad93ec9","entry":"Model","repo":"ikostrikov/jax-rl","repo_kind":"listed","path":"jaxrl/agents/sac/sac_learner.py","file_url":"https://github.com/ikostrikov/jax-rl/blob/HEAD/jaxrl/agents/sac/sac_learner.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"3ef56075bad93ec9"}},{"code_sha256_prefix":"7524ef9f7598dd6f","entry":"OBAC","repo":"roythuly/obac","repo_kind":"listed","path":"model/algo.py","file_url":"https://github.com/roythuly/obac/blob/HEAD/model/algo.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"7524ef9f7598dd6f"}},{"code_sha256_prefix":"b36d35fd411b0150","entry":"RLNetwork","repo":"kushagra06/SAC","repo_kind":"listed","path":"model.py","file_url":"https://github.com/kushagra06/SAC/blob/HEAD/model.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"b36d35fd411b0150"}},{"code_sha256_prefix":"995a5d6382ee476d","entry":"ReplayMemory","repo":"ajaysub110/rl-pytorch","repo_kind":"listed","path":"sac/models/soft_actor_critic.py","file_url":"https://github.com/ajaysub110/rl-pytorch/blob/HEAD/sac/models/soft_actor_critic.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"995a5d6382ee476d"}},{"code_sha256_prefix":"569d921eb9b51536","entry":"SAC","repo":"kdally/fault-tolerant-flight-control-drl","repo_kind":"listed","path":"fault_tolerant_flight_control_drl/agent/sac.py","file_url":"https://github.com/kdally/fault-tolerant-flight-control-drl/blob/HEAD/fault_tolerant_flight_control_drl/agent/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"569d921eb9b51536"}},{"code_sha256_prefix":"0edd4b0f90be8f84","entry":"SAC","repo":"baturaysaglam/la3p","repo_kind":"listed","path":"Code/SAC/SAC.py","file_url":"https://github.com/baturaysaglam/la3p/blob/HEAD/Code/SAC/SAC.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"0edd4b0f90be8f84"}},{"code_sha256_prefix":"4849ba61ae47dd70","entry":"SAC","repo":"yhisaki/average-reward-drl","repo_kind":"listed","path":"average_reward_drl/algorithms/sac.py","file_url":"https://github.com/yhisaki/average-reward-drl/blob/HEAD/average_reward_drl/algorithms/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"4849ba61ae47dd70"}},{"code_sha256_prefix":"348b1924f106de50","entry":"SAC","repo":"ku2482/discor.pytorch","repo_kind":"listed","path":"discor/algorithm/sac.py","file_url":"https://github.com/ku2482/discor.pytorch/blob/HEAD/discor/algorithm/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"348b1924f106de50"}},{"code_sha256_prefix":"9e2e2c85eabbd88f","entry":"SAC","repo":"h-aboutalebi/SparceReward","repo_kind":"listed","path":"engine/algorithms/SAC/sac.py","file_url":"https://github.com/h-aboutalebi/SparceReward/blob/HEAD/engine/algorithms/SAC/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"9e2e2c85eabbd88f"}},{"code_sha256_prefix":"1cb5a8aa4ee6cb43","entry":"SAC","repo":"sunfex/weighted-sac","repo_kind":"listed","path":"sac.py","file_url":"https://github.com/sunfex/weighted-sac/blob/HEAD/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"1cb5a8aa4ee6cb43"}},{"code_sha256_prefix":"9efbd1e63c6644e5","entry":"SAC","repo":"hyunin-lee/ForecasterSAC","repo_kind":"listed","path":"sac.py","file_url":"https://github.com/hyunin-lee/ForecasterSAC/blob/HEAD/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"9efbd1e63c6644e5"}},{"code_sha256_prefix":"6af70c8a2239d7c5","entry":"SAC","repo":"garyzyr001/rethinking-airl","repo_kind":"listed","path":"airl/algo/sac.py","file_url":"https://github.com/garyzyr001/rethinking-airl/blob/HEAD/airl/algo/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"6af70c8a2239d7c5"}},{"code_sha256_prefix":"fa07dbb92f6c222b","entry":"SAC","repo":"MarsEleven/car_racer_RL","repo_kind":"listed","path":"sac/sac.py","file_url":"https://github.com/MarsEleven/car_racer_RL/blob/HEAD/sac/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"fa07dbb92f6c222b"}},{"code_sha256_prefix":"104ef2d3e37baa21","entry":"SAC","repo":"ku2482/gail-airl-ppo.pytorch","repo_kind":"listed","path":"gail_airl_ppo/algo/sac.py","file_url":"https://github.com/ku2482/gail-airl-ppo.pytorch/blob/HEAD/gail_airl_ppo/algo/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"104ef2d3e37baa21"}},{"code_sha256_prefix":"d77f61cb28da8383","entry":"SAC","repo":"pranz24/pytorch-soft-actor-critic","repo_kind":"listed","path":"sac.py","file_url":"https://github.com/pranz24/pytorch-soft-actor-critic/blob/HEAD/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"d77f61cb28da8383"}},{"code_sha256_prefix":"6bc96035de025e83","entry":"SACLearner","repo":"ikostrikov/jax-rl","repo_kind":"listed","path":"jaxrl/agents/sac/sac_learner.py","file_url":"https://github.com/ikostrikov/jax-rl/blob/HEAD/jaxrl/agents/sac/sac_learner.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"6bc96035de025e83"}},{"code_sha256_prefix":"3e1e70dd1d5f3958","entry":"SAC_Trainer","repo":"quantumiracle/Popular-RL-Algorithms","repo_kind":"listed","path":"sac_v2.py","file_url":"https://github.com/quantumiracle/Popular-RL-Algorithms/blob/HEAD/sac_v2.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"3e1e70dd1d5f3958"}},{"code_sha256_prefix":"ecf34b65e8d9d3bb","entry":"SoftActorCritic","repo":"ajaysub110/rl-pytorch","repo_kind":"listed","path":"sac/models/soft_actor_critic.py","file_url":"https://github.com/ajaysub110/rl-pytorch/blob/HEAD/sac/models/soft_actor_critic.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"ecf34b65e8d9d3bb"}},{"code_sha256_prefix":"231f633be51bd617","entry":"_soft_update","repo":"ku2482/discor.pytorch","repo_kind":"listed","path":"discor/algorithm/sac.py","file_url":"https://github.com/ku2482/discor.pytorch/blob/HEAD/discor/algorithm/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"231f633be51bd617"}},{"code_sha256_prefix":"f07528bc5e071757","entry":"_update_jit","repo":"ikostrikov/jax-rl","repo_kind":"listed","path":"jaxrl/agents/sac/sac_learner.py","file_url":"https://github.com/ikostrikov/jax-rl/blob/HEAD/jaxrl/agents/sac/sac_learner.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"f07528bc5e071757"}},{"code_sha256_prefix":"198be549ce9696ef","entry":"apply_squashing_func","repo":"watchernyu/spinningup-drl-prototyping","repo_kind":"listed","path":"spinup/algos/sac/core.py","file_url":"https://github.com/watchernyu/spinningup-drl-prototyping/blob/HEAD/spinup/algos/sac/core.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"198be549ce9696ef"}},{"code_sha256_prefix":"ded76544fe9d6607","entry":"assert_action","repo":"ku2482/discor.pytorch","repo_kind":"listed","path":"discor/algorithm/sac.py","file_url":"https://github.com/ku2482/discor.pytorch/blob/HEAD/discor/algorithm/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"ded76544fe9d6607"}},{"code_sha256_prefix":"d1076220e33b55ba","entry":"copy_params","repo":"ajaysub110/rl-pytorch","repo_kind":"listed","path":"sac/models/soft_actor_critic.py","file_url":"https://github.com/ajaysub110/rl-pytorch/blob/HEAD/sac/models/soft_actor_critic.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"d1076220e33b55ba"}},{"code_sha256_prefix":"ed542fae5c7f2c04","entry":"disable_gradient","repo":"garyzyr001/rethinking-airl","repo_kind":"listed","path":"airl/algo/sac.py","file_url":"https://github.com/garyzyr001/rethinking-airl/blob/HEAD/airl/algo/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"ed542fae5c7f2c04"}},{"code_sha256_prefix":"799c08d64f58d9ef","entry":"hard_update","repo":"Rafael1s/Deep-Reinforcement-Learning-Udacity","repo_kind":"listed","path":"Ant-PyBulletEnv-Soft-Actor-Critic/sac_agent.py","file_url":"https://github.com/Rafael1s/Deep-Reinforcement-Learning-Udacity/blob/HEAD/Ant-PyBulletEnv-Soft-Actor-Critic/sac_agent.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"799c08d64f58d9ef"}},{"code_sha256_prefix":"7abe222eb3afeb6f","entry":"hard_update","repo":"ShawK91/Evolutionary-Reinforcement-Learning","repo_kind":"listed","path":"algos/sac.py","file_url":"https://github.com/ShawK91/Evolutionary-Reinforcement-Learning/blob/HEAD/algos/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"7abe222eb3afeb6f"}},{"code_sha256_prefix":"1adf91cc413cf6c2","entry":"hard_update","repo":"MarsEleven/car_racer_RL","repo_kind":"listed","path":"sac/sac.py","file_url":"https://github.com/MarsEleven/car_racer_RL/blob/HEAD/sac/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"1adf91cc413cf6c2"}},{"code_sha256_prefix":"8576d4b41e6ed12f","entry":"init_weights","repo":"ajaysub110/rl-pytorch","repo_kind":"listed","path":"sac/models/soft_actor_critic.py","file_url":"https://github.com/ajaysub110/rl-pytorch/blob/HEAD/sac/models/soft_actor_critic.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"8576d4b41e6ed12f"}},{"code_sha256_prefix":"544fe1f08a36cc62","entry":"initialize_weights_xavier","repo":"ku2482/discor.pytorch","repo_kind":"listed","path":"discor/algorithm/sac.py","file_url":"https://github.com/ku2482/discor.pytorch/blob/HEAD/discor/algorithm/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"544fe1f08a36cc62"}},{"code_sha256_prefix":"706459e26fbb0b4c","entry":"load_model","repo":"thomashirtz/soft-actor-critic","repo_kind":"listed","path":"soft_actor_critic/agent.py","file_url":"https://github.com/thomashirtz/soft-actor-critic/blob/HEAD/soft_actor_critic/agent.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"706459e26fbb0b4c"}},{"code_sha256_prefix":"e68b07bcdded00fa","entry":"mlp","repo":"watchernyu/spinningup-drl-prototyping","repo_kind":"listed","path":"spinup/algos/sac/core.py","file_url":"https://github.com/watchernyu/spinningup-drl-prototyping/blob/HEAD/spinup/algos/sac/core.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"e68b07bcdded00fa"}},{"code_sha256_prefix":"8498689f07030ac8","entry":"mlp_actor_critic","repo":"watchernyu/spinningup-drl-prototyping","repo_kind":"listed","path":"spinup/algos/sac/core.py","file_url":"https://github.com/watchernyu/spinningup-drl-prototyping/blob/HEAD/spinup/algos/sac/core.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"8498689f07030ac8"}},{"code_sha256_prefix":"a860fee8188a0218","entry":"mlp_gaussian_policy","repo":"watchernyu/spinningup-drl-prototyping","repo_kind":"listed","path":"spinup/algos/sac/core.py","file_url":"https://github.com/watchernyu/spinningup-drl-prototyping/blob/HEAD/spinup/algos/sac/core.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"a860fee8188a0218"}},{"code_sha256_prefix":"f299a2900b4034fa","entry":"polyak_update","repo":"yhisaki/average-reward-drl","repo_kind":"listed","path":"average_reward_drl/algorithms/sac.py","file_url":"https://github.com/yhisaki/average-reward-drl/blob/HEAD/average_reward_drl/algorithms/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"f299a2900b4034fa"}},{"code_sha256_prefix":"7d297778525c40d1","entry":"save_model","repo":"thomashirtz/soft-actor-critic","repo_kind":"listed","path":"soft_actor_critic/agent.py","file_url":"https://github.com/thomashirtz/soft-actor-critic/blob/HEAD/soft_actor_critic/agent.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"7d297778525c40d1"}},{"code_sha256_prefix":"0dfad75cd2d202d6","entry":"set_global_seeds","repo":"kdally/fault-tolerant-flight-control-drl","repo_kind":"listed","path":"fault_tolerant_flight_control_drl/agent/sac.py","file_url":"https://github.com/kdally/fault-tolerant-flight-control-drl/blob/HEAD/fault_tolerant_flight_control_drl/agent/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"0dfad75cd2d202d6"}},{"code_sha256_prefix":"f735812881238a1f","entry":"soft_update","repo":"Rafael1s/Deep-Reinforcement-Learning-Udacity","repo_kind":"listed","path":"Ant-PyBulletEnv-Soft-Actor-Critic/sac_agent.py","file_url":"https://github.com/Rafael1s/Deep-Reinforcement-Learning-Udacity/blob/HEAD/Ant-PyBulletEnv-Soft-Actor-Critic/sac_agent.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"f735812881238a1f"}},{"code_sha256_prefix":"e8abbe917721f0c8","entry":"soft_update","repo":"ku2482/discor.pytorch","repo_kind":"listed","path":"discor/algorithm/sac.py","file_url":"https://github.com/ku2482/discor.pytorch/blob/HEAD/discor/algorithm/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"e8abbe917721f0c8"}},{"code_sha256_prefix":"fcdeb82694a504e1","entry":"soft_update","repo":"ShawK91/Evolutionary-Reinforcement-Learning","repo_kind":"listed","path":"algos/sac.py","file_url":"https://github.com/ShawK91/Evolutionary-Reinforcement-Learning/blob/HEAD/algos/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"fcdeb82694a504e1"}},{"code_sha256_prefix":"8dc0540b8dfe9b97","entry":"soft_update","repo":"garyzyr001/rethinking-airl","repo_kind":"listed","path":"airl/algo/sac.py","file_url":"https://github.com/garyzyr001/rethinking-airl/blob/HEAD/airl/algo/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"8dc0540b8dfe9b97"}},{"code_sha256_prefix":"ace24484ab1d6b7f","entry":"soft_update","repo":"MarsEleven/car_racer_RL","repo_kind":"listed","path":"sac/sac.py","file_url":"https://github.com/MarsEleven/car_racer_RL/blob/HEAD/sac/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"ace24484ab1d6b7f"}},{"code_sha256_prefix":"54ab8d446018b958","entry":"soft_update","repo":"ajaysub110/rl-pytorch","repo_kind":"listed","path":"sac/models/soft_actor_critic.py","file_url":"https://github.com/ajaysub110/rl-pytorch/blob/HEAD/sac/models/soft_actor_critic.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"54ab8d446018b958"}},{"code_sha256_prefix":"f62b5d3c6983e5aa","entry":"target_update","repo":"ikostrikov/jax-rl","repo_kind":"listed","path":"jaxrl/agents/sac/sac_learner.py","file_url":"https://github.com/ikostrikov/jax-rl/blob/HEAD/jaxrl/agents/sac/sac_learner.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"f62b5d3c6983e5aa"}},{"code_sha256_prefix":"85fff8ae028fc1dd","entry":"update_network_parameters","repo":"thomashirtz/soft-actor-critic","repo_kind":"listed","path":"soft_actor_critic/agent.py","file_url":"https://github.com/thomashirtz/soft-actor-critic/blob/HEAD/soft_actor_critic/agent.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"85fff8ae028fc1dd"}},{"code_sha256_prefix":"68c08ef9d5d473a7","entry":"weight_init","repo":"SaminYeasar/off_policy_ac","repo_kind":"listed","path":"SAC/SAC.py","file_url":"https://github.com/SaminYeasar/off_policy_ac/blob/HEAD/SAC/SAC.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"68c08ef9d5d473a7"}},{"code_sha256_prefix":"6ae1b395aeb4e7d3","entry":"weight_initialization","repo":"thomashirtz/soft-actor-critic","repo_kind":"listed","path":"soft_actor_critic/agent.py","file_url":"https://github.com/thomashirtz/soft-actor-critic/blob/HEAD/soft_actor_critic/agent.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"6ae1b395aeb4e7d3"}},{"code_sha256_prefix":"0537c4bc8300d410","entry":"weights_init_","repo":"Rafael1s/Deep-Reinforcement-Learning-Udacity","repo_kind":"listed","path":"Ant-PyBulletEnv-Soft-Actor-Critic/sac_agent.py","file_url":"https://github.com/Rafael1s/Deep-Reinforcement-Learning-Udacity/blob/HEAD/Ant-PyBulletEnv-Soft-Actor-Critic/sac_agent.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"0537c4bc8300d410"}},{"code_sha256_prefix":"abe2b5a0b5e817fd","entry":"weights_init_","repo":"MarsEleven/car_racer_RL","repo_kind":"listed","path":"sac/sac.py","file_url":"https://github.com/MarsEleven/car_racer_RL/blob/HEAD/sac/sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"abe2b5a0b5e817fd"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}