{"url":"/task/smac-1","name":"SMAC+","slug":"smac-1","description_markdown":"Bechmarks for Efficient Exploration of Completion of Multi-stage Tasks and Usage of Environmental Factors","categories":[{"name":"Playing Games","url":"/area/playing-games"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":126,"papers_with_code":58,"benchmarks":16,"benchmark_tables_in_archive":16,"benchmark_tables_shown":16,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":17,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/smac-on-smac-def-armored-sequential","slug":"smac-on-smac-def-armored-sequential","dataset":"Def_Armored_sequential","dataset_url":"/dataset/smac-def-armored-sequential","rows_in_archive":11,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"DRIMA","paper_title":"Disentangling Sources of Risk for Distributional Multi-Agent Reinforcement Learning","paper_url":"/paper/disentangling-sources-of-risk-for","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-def-infantry-sequential","slug":"smac-on-smac-def-infantry-sequential","dataset":"Def_Infantry_sequential","dataset_url":"/dataset/smac-def-infantry-sequential","rows_in_archive":11,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"MADDPG","paper_title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","paper_url":"/paper/multi-agent-actor-critic-for-mixed","paper_date":"2017-06-07","arxiv_id":"1706.02275","code_links":[{"title":"ray-project/ray","url":"https://github.com/ray-project/ray/tree/master/rllib"},{"title":"openai/multiagent-particle-envs","url":"https://github.com/openai/multiagent-particle-envs"},{"title":"openai/maddpg","url":"https://github.com/openai/maddpg"},{"title":"xuehy/pytorch-maddpg","url":"https://github.com/xuehy/pytorch-maddpg"},{"title":"shariqiqbal2810/maddpg-pytorch","url":"https://github.com/shariqiqbal2810/maddpg-pytorch"},{"title":"starry-sky6688/MADDPG","url":"https://github.com/starry-sky6688/MADDPG"},{"title":"facebookresearch/benchmarl","url":"https://github.com/facebookresearch/benchmarl"},{"title":"philtabor/Multi-Agent-Deep-Deterministic-Policy-Gradients","url":"https://github.com/philtabor/Multi-Agent-Deep-Deterministic-Policy-Gradients"},{"title":"cyanrain7/trpo-in-marl","url":"https://github.com/cyanrain7/trpo-in-marl"},{"title":"cyanrain7/trust-region-policy-optimisation-in-multi-agent-reinforcement-learning","url":"https://github.com/cyanrain7/trust-region-policy-optimisation-in-multi-agent-reinforcement-learning"},{"title":"JohannesAck/tf2multiagentrl","url":"https://github.com/JohannesAck/tf2multiagentrl"},{"title":"isp1tze/MAProj","url":"https://github.com/isp1tze/MAProj"},{"title":"JohannesAck/MATD3implementation","url":"https://github.com/JohannesAck/MATD3implementation"},{"title":"thechrisyoon08/Multi-agent-reinforcement-learning","url":"https://github.com/thechrisyoon08/Multi-agent-reinforcement-learning"},{"title":"thechrisyoon08/marl","url":"https://github.com/thechrisyoon08/marl"},{"title":"cyoon1729/Multi-agent-reinforcement-learning","url":"https://github.com/cyoon1729/Multi-agent-reinforcement-learning"},{"title":"quantumiracle/mars","url":"https://github.com/quantumiracle/mars"},{"title":"shariqiqbal2810/multiagent-particle-envs","url":"https://github.com/shariqiqbal2810/multiagent-particle-envs"},{"title":"morning9393/HAPPO-HATRPO","url":"https://github.com/morning9393/HAPPO-HATRPO"},{"title":"google/maddpg-replication","url":"https://github.com/google/maddpg-replication"},{"title":"caslab-vt/SARNet","url":"https://github.com/caslab-vt/SARNet"},{"title":"qi-pang/mdpfuzz","url":"https://github.com/qi-pang/mdpfuzz"},{"title":"pr-shukla/maddpg-keras","url":"https://github.com/pr-shukla/maddpg-keras"},{"title":"gingkg/multiagent-particle-envs","url":"https://github.com/gingkg/multiagent-particle-envs"},{"title":"jyqhahah/rl_maddpg_matd3","url":"https://github.com/jyqhahah/rl_maddpg_matd3"},{"title":"baoqianwang/iros22_darl1n","url":"https://github.com/baoqianwang/iros22_darl1n"},{"title":"zowiezhang/stas","url":"https://github.com/zowiezhang/stas"},{"title":"yannbouteiller/gym-airsimdroneracinglab","url":"https://github.com/yannbouteiller/gym-airsimdroneracinglab"},{"title":"anonymous-iclr22/trust-region-in-multi-agent-reinforcement-learning","url":"https://github.com/anonymous-iclr22/trust-region-in-multi-agent-reinforcement-learning"},{"title":"schroederdewitt/multiagent-particle-envs","url":"https://github.com/schroederdewitt/multiagent-particle-envs"},{"title":"EyaRhouma/collaboration-competition-MADDPG","url":"https://github.com/EyaRhouma/collaboration-competition-MADDPG"},{"title":"zoeyuchao/MPE-pytorch","url":"https://github.com/zoeyuchao/MPE-pytorch"},{"title":"bonniesjli/MADDPG_Tennis","url":"https://github.com/bonniesjli/MADDPG_Tennis"},{"title":"bonniesjli/MADDPG_Tennis_UnityML","url":"https://github.com/bonniesjli/MADDPG_Tennis_UnityML"},{"title":"semitable/multiagent-particle-envs","url":"https://github.com/semitable/multiagent-particle-envs"},{"title":"kargarisaac/macrpo","url":"https://github.com/kargarisaac/macrpo"},{"title":"madras-simulator/Multi-Agent-Particle-Environment","url":"https://github.com/madras-simulator/Multi-Agent-Particle-Environment"},{"title":"wsjeon/multiagent-particle-envs-maac","url":"https://github.com/wsjeon/multiagent-particle-envs-maac"},{"title":"NeuroCSUT/intentions","url":"https://github.com/NeuroCSUT/intentions"},{"title":"marwanihab/RL_Tag_Game","url":"https://github.com/marwanihab/RL_Tag_Game"},{"title":"zoeyuchao/MPEnew-pytorch","url":"https://github.com/zoeyuchao/MPEnew-pytorch"},{"title":"lachisis/multiagent-particle-envs","url":"https://github.com/lachisis/multiagent-particle-envs"},{"title":"Stippler/cow-simulator","url":"https://github.com/Stippler/cow-simulator"},{"title":"biorobotics/PRD_environments","url":"https://github.com/biorobotics/PRD_environments"},{"title":"raoshashank/Tennis-with-MADDPG","url":"https://github.com/raoshashank/Tennis-with-MADDPG"},{"title":"Steven-Ho/multiagent-particle-envs","url":"https://github.com/Steven-Ho/multiagent-particle-envs"},{"title":"sliao-mi-luku/DeepRL-multiple-agents-tennis-udacity-drlnd-p3","url":"https://github.com/sliao-mi-luku/DeepRL-multiple-agents-tennis-udacity-drlnd-p3"},{"title":"Ah31/maddpg_pytorch","url":"https://github.com/Ah31/maddpg_pytorch"},{"title":"karishnu/tf-agents-multi-particle-envs","url":"https://github.com/karishnu/tf-agents-multi-particle-envs"},{"title":"SihongHo/multiagent-particle-envs","url":"https://github.com/SihongHo/multiagent-particle-envs"},{"title":"wsjeon/multiagent-particle-envs-v2","url":"https://github.com/wsjeon/multiagent-particle-envs-v2"},{"title":"baicenxiao/shaping-advice","url":"https://github.com/baicenxiao/shaping-advice"},{"title":"Yutongamber/MADDPG","url":"https://github.com/Yutongamber/MADDPG"},{"title":"SintolRTOS/multi-agent_Example","url":"https://github.com/SintolRTOS/multi-agent_Example"},{"title":"Chan1998/MAAC","url":"https://github.com/Chan1998/MAAC"},{"title":"Abdelhamid-bouzid/Multi-Agent-Deep-Deterministic-Policy-Gradient","url":"https://github.com/Abdelhamid-bouzid/Multi-Agent-Deep-Deterministic-Policy-Gradient"},{"title":"debajit15kgp/multiagent-envs","url":"https://github.com/debajit15kgp/multiagent-envs"},{"title":"johannesharmse/multi_agent_RL","url":"https://github.com/johannesharmse/multi_agent_RL"},{"title":"jingdic/rgmcomm","url":"https://github.com/jingdic/rgmcomm"},{"title":"rainandwind1/Maddpg_multiagent","url":"https://github.com/rainandwind1/Maddpg_multiagent"},{"title":"kishorkuttan/Multi-Agent-Deep-Deterministic-Policy-Gradient-actor-critic-deep-reinforcement-for-capacity-manageme","url":"https://github.com/kishorkuttan/Multi-Agent-Deep-Deterministic-Policy-Gradient-actor-critic-deep-reinforcement-for-capacity-manageme"},{"title":"dtabas/multiagent-particle-envs","url":"https://github.com/dtabas/multiagent-particle-envs"},{"title":"goldbattle/snakes_mal","url":"https://github.com/goldbattle/snakes_mal"},{"title":"ksajan/DDPG-MAPE","url":"https://github.com/ksajan/DDPG-MAPE"},{"title":"rainandwind1/MERL","url":"https://github.com/rainandwind1/MERL"},{"title":"darshil333/CSE574","url":"https://github.com/darshil333/CSE574"},{"title":"rallen10/multiagent-particle-envs","url":"https://github.com/rallen10/multiagent-particle-envs"},{"title":"tkarr21/multagent-particle-envs","url":"https://github.com/tkarr21/multagent-particle-envs"},{"title":"mauricemager/multiagent-robot","url":"https://github.com/mauricemager/multiagent-robot"},{"title":"tkarr21/multiagent-particle-envs","url":"https://github.com/tkarr21/multiagent-particle-envs"},{"title":"petsol/MultiAgentCooperation_UnityAgent_MADDPG_Udacity","url":"https://github.com/petsol/MultiAgentCooperation_UnityAgent_MADDPG_Udacity"},{"title":"MrDaubinet/collaboration-and-competition","url":"https://github.com/MrDaubinet/collaboration-and-competition"},{"title":"LXYYY/multiagent-particle-envs","url":"https://github.com/LXYYY/multiagent-particle-envs"},{"title":"Zorrorulz/MultiAgentDDPG-Tennis","url":"https://github.com/Zorrorulz/MultiAgentDDPG-Tennis"},{"title":"biemann/Collaboration-and-Competition","url":"https://github.com/biemann/Collaboration-and-Competition"},{"title":"hepengli/multiagent-particle-envs","url":"https://github.com/hepengli/multiagent-particle-envs"},{"title":"jiayu-ch15/MPE-for-curriculum-learning","url":"https://github.com/jiayu-ch15/MPE-for-curriculum-learning"},{"title":"marwanihab/RL_Testing_Noise_ASRN","url":"https://github.com/marwanihab/RL_Testing_Noise_ASRN"},{"title":"AleXander-Tsui/MPE","url":"https://github.com/AleXander-Tsui/MPE"},{"title":"madhur-tandon/RL-Project","url":"https://github.com/madhur-tandon/RL-Project"},{"title":"rainandwind1/MADDPG-reconstruct","url":"https://github.com/rainandwind1/MADDPG-reconstruct"},{"title":"RL-WFU/multi_agent_attack","url":"https://github.com/RL-WFU/multi_agent_attack"},{"title":"JinTanda/MADDPG_env","url":"https://github.com/JinTanda/MADDPG_env"},{"title":"jansenkeith501/CS295-MADDPG","url":"https://github.com/jansenkeith501/CS295-MADDPG"},{"title":"baradist/multiagent-particle-envs","url":"https://github.com/baradist/multiagent-particle-envs"},{"title":"krasing/DRLearningCollaboration","url":"https://github.com/krasing/DRLearningCollaboration"}],"syntology":{"n":143,"n_ran":75,"n_unverified":68,"n_pointer_only":99}}},{"leaderboard":"/sota/smac-on-smac-def-outnumbered-sequential","slug":"smac-on-smac-def-outnumbered-sequential","dataset":"Def_Outnumbered_sequential","dataset_url":"/dataset/smac-def-outnumbered-sequential","rows_in_archive":11,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"DRIMA","paper_title":"Disentangling Sources of Risk for Distributional Multi-Agent Reinforcement Learning","paper_url":"/paper/disentangling-sources-of-risk-for","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-def-armored-parallel","slug":"smac-on-smac-def-armored-parallel","dataset":"Def_Armored_parallel","dataset_url":"/dataset/smac-def-armored-parallel","rows_in_archive":10,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"DMIX","paper_title":"DFAC Framework: Factorizing the Value Function via Quantile Mixture for Multi-Agent Distributional Q-Learning","paper_url":"/paper/dfac-framework-factorizing-the-value-function","paper_date":"2021-02-16","arxiv_id":"2102.07936","code_links":[{"title":"j3soon/dfac","url":"https://github.com/j3soon/dfac"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/smac-on-smac-def-infantry-parallel","slug":"smac-on-smac-def-infantry-parallel","dataset":"Def_Infantry_parallel","dataset_url":"/dataset/smac-def-infantry-parallel","rows_in_archive":10,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"QTRAN","paper_title":"QTRAN: Learning to Factorize with Transformation for Cooperative Multi-Agent Reinforcement Learning","paper_url":"/paper/qtran-learning-to-factorize-with","paper_date":"2019-05-14","arxiv_id":"1905.05408","code_links":[{"title":"opendilab/DI-engine","url":"https://github.com/opendilab/DI-engine/blob/main/ding/policy/qtran.py"},{"title":"hhhusiyi-monash/UPDeT","url":"https://github.com/hhhusiyi-monash/UPDeT"},{"title":"Sonkyunghwan/QTRAN","url":"https://github.com/Sonkyunghwan/QTRAN"},{"title":"jugg1er/air","url":"https://github.com/jugg1er/air"}],"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}}},{"leaderboard":"/sota/smac-on-smac-def-outnumbered-parallel","slug":"smac-on-smac-def-outnumbered-parallel","dataset":"Def_Outnumbered_parallel","dataset_url":"/dataset/smac-def-outnumbered-parallel","rows_in_archive":10,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"DRIMA","paper_title":"Disentangling Sources of Risk for Distributional Multi-Agent Reinforcement Learning","paper_url":"/paper/disentangling-sources-of-risk-for","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-off-complicated-parallel","slug":"smac-on-smac-off-complicated-parallel","dataset":"Off_Complicated_parallel","dataset_url":"/dataset/smac-off-complicated-parallel","rows_in_archive":10,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"DRIMA","paper_title":"Disentangling Sources of Risk for Distributional Multi-Agent Reinforcement Learning","paper_url":"/paper/disentangling-sources-of-risk-for","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-off-distant-parallel","slug":"smac-on-smac-off-distant-parallel","dataset":"Off_Distant_parallel","dataset_url":"/dataset/smac-off-distant-parallel","rows_in_archive":10,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"DRIMA","paper_title":"Disentangling Sources of Risk for Distributional Multi-Agent Reinforcement Learning","paper_url":"/paper/disentangling-sources-of-risk-for","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-off-hard-parallel","slug":"smac-on-smac-off-hard-parallel","dataset":"Off_Hard_parallel","dataset_url":"/dataset/smac-off-hard-parallel","rows_in_archive":10,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"DRIMA","paper_title":"Disentangling Sources of Risk for Distributional Multi-Agent Reinforcement Learning","paper_url":"/paper/disentangling-sources-of-risk-for","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-off-near-parallel","slug":"smac-on-smac-off-near-parallel","dataset":"Off_Near_parallel","dataset_url":"/dataset/smac-off-near-parallel","rows_in_archive":10,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"QMIX","paper_title":"QMIX: Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","paper_url":"/paper/qmix-monotonic-value-function-factorisation","paper_date":"2018-03-30","arxiv_id":"1803.11485","code_links":[{"title":"ray-project/ray","url":"https://github.com/ray-project/ray/tree/master/rllib"},{"title":"opendilab/DI-engine","url":"https://github.com/opendilab/DI-engine/blob/main/ding/policy/qmix.py"},{"title":"oxwhirl/pymarl","url":"https://github.com/oxwhirl/pymarl"},{"title":"starry-sky6688/marl-algorithms","url":"https://github.com/starry-sky6688/marl-algorithms"},{"title":"oxwhirl/smac","url":"https://github.com/oxwhirl/smac"},{"title":"facebookresearch/benchmarl","url":"https://github.com/facebookresearch/benchmarl"},{"title":"hhhusiyi-monash/UPDeT","url":"https://github.com/hhhusiyi-monash/UPDeT"},{"title":"TonghanWang/NDQ","url":"https://github.com/TonghanWang/NDQ"},{"title":"nju-rl/acorm","url":"https://github.com/nju-rl/acorm"},{"title":"TonghanWang/DOP","url":"https://github.com/TonghanWang/DOP"},{"title":"puyuan1996/MARL","url":"https://github.com/puyuan1996/MARL"},{"title":"xiuyu0000/new_papers_codes","url":"https://github.com/xiuyu0000/new_papers_codes/tree/main/qmix"},{"title":"cathyhxh/ctds","url":"https://github.com/cathyhxh/ctds"},{"title":"jugg1er/air","url":"https://github.com/jugg1er/air"},{"title":"ifpen/wfcrl-benchmark","url":"https://github.com/ifpen/wfcrl-benchmark"},{"title":"gingkg/smac","url":"https://github.com/gingkg/smac"},{"title":"15534081591/QMIX","url":"https://github.com/15534081591/QMIX"},{"title":"2023-MindSpore-1/ms-code-221","url":"https://github.com/2023-MindSpore-1/ms-code-221/tree/main/qmix"}],"syntology":{"n":11,"n_ran":9,"n_unverified":2,"n_pointer_only":6}}},{"leaderboard":"/sota/smac-on-smac-off-superhard-parallel","slug":"smac-on-smac-off-superhard-parallel","dataset":"Off_Superhard_parallel","dataset_url":"/dataset/smac-off-superhard-parallel","rows_in_archive":10,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"MASAC","paper_title":"Decomposed Soft Actor-Critic Method for Cooperative Multi-Agent Reinforcement Learning","paper_url":"/paper/decomposed-soft-actor-critic-method-for","paper_date":"2021-04-14","arxiv_id":"2104.06655","code_links":[{"title":"puyuan1996/MARL","url":"https://github.com/puyuan1996/MARL"}],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-off-complicated-sequential","slug":"smac-on-smac-off-complicated-sequential","dataset":"Off_Complicated_sequential","dataset_url":"/dataset/smac-off-complicated-sequential","rows_in_archive":4,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"DRIMA","paper_title":"Disentangling Sources of Risk for Distributional Multi-Agent Reinforcement Learning","paper_url":"/paper/disentangling-sources-of-risk-for","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-off-distant-sequential","slug":"smac-on-smac-off-distant-sequential","dataset":"Off_Distant_sequential","dataset_url":"/dataset/smac-off-distant-sequential","rows_in_archive":4,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"DRIMA","paper_title":"Disentangling Sources of Risk for Distributional Multi-Agent Reinforcement Learning","paper_url":"/paper/disentangling-sources-of-risk-for","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-off-hard-sequential","slug":"smac-on-smac-off-hard-sequential","dataset":"Off_Hard_sequential","dataset_url":"/dataset/smac-off-hard-sequential","rows_in_archive":4,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"QMIX","paper_title":"QMIX: Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","paper_url":"/paper/qmix-monotonic-value-function-factorisation","paper_date":"2018-03-30","arxiv_id":"1803.11485","code_links":[{"title":"ray-project/ray","url":"https://github.com/ray-project/ray/tree/master/rllib"},{"title":"opendilab/DI-engine","url":"https://github.com/opendilab/DI-engine/blob/main/ding/policy/qmix.py"},{"title":"oxwhirl/pymarl","url":"https://github.com/oxwhirl/pymarl"},{"title":"starry-sky6688/marl-algorithms","url":"https://github.com/starry-sky6688/marl-algorithms"},{"title":"oxwhirl/smac","url":"https://github.com/oxwhirl/smac"},{"title":"facebookresearch/benchmarl","url":"https://github.com/facebookresearch/benchmarl"},{"title":"hhhusiyi-monash/UPDeT","url":"https://github.com/hhhusiyi-monash/UPDeT"},{"title":"TonghanWang/NDQ","url":"https://github.com/TonghanWang/NDQ"},{"title":"nju-rl/acorm","url":"https://github.com/nju-rl/acorm"},{"title":"TonghanWang/DOP","url":"https://github.com/TonghanWang/DOP"},{"title":"puyuan1996/MARL","url":"https://github.com/puyuan1996/MARL"},{"title":"xiuyu0000/new_papers_codes","url":"https://github.com/xiuyu0000/new_papers_codes/tree/main/qmix"},{"title":"cathyhxh/ctds","url":"https://github.com/cathyhxh/ctds"},{"title":"jugg1er/air","url":"https://github.com/jugg1er/air"},{"title":"ifpen/wfcrl-benchmark","url":"https://github.com/ifpen/wfcrl-benchmark"},{"title":"gingkg/smac","url":"https://github.com/gingkg/smac"},{"title":"15534081591/QMIX","url":"https://github.com/15534081591/QMIX"},{"title":"2023-MindSpore-1/ms-code-221","url":"https://github.com/2023-MindSpore-1/ms-code-221/tree/main/qmix"}],"syntology":{"n":11,"n_ran":9,"n_unverified":2,"n_pointer_only":6}}},{"leaderboard":"/sota/smac-on-smac-off-near-sequential","slug":"smac-on-smac-off-near-sequential","dataset":"Off_Near_sequential","dataset_url":"/dataset/smac-off-near-sequential","rows_in_archive":4,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"DRIMA","paper_title":"Disentangling Sources of Risk for Distributional Multi-Agent Reinforcement Learning","paper_url":"/paper/disentangling-sources-of-risk-for","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-off-superhard-sequential","slug":"smac-on-smac-off-superhard-sequential","dataset":"Off_Superhard_sequential","dataset_url":"/dataset/smac-off-superhard-sequential","rows_in_archive":4,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"DRIMA","paper_title":"Disentangling Sources of Risk for Distributional Multi-Agent Reinforcement Learning","paper_url":"/paper/disentangling-sources-of-risk-for","paper_date":"2021-09-29","arxiv_id":null,"code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/smac-plus","name":"SMAC-Exp","full_name":"StarCraft Multi-Agent Exploration Challenge","num_papers_in_archive":11},{"url":"/dataset/smac-def-armored-sequential","name":"Def_Armored_sequential","full_name":"","num_papers_in_archive":9},{"url":"/dataset/smac-def-infantry-parallel","name":"Def_Infantry_parallel","full_name":"SMAC+_Def_Infantry_parallel_20","num_papers_in_archive":9},{"url":"/dataset/smac-def-infantry-sequential","name":"Def_Infantry_sequential","full_name":"","num_papers_in_archive":9},{"url":"/dataset/smac-def-outnumbered-sequential","name":"Def_Outnumbered_sequential","full_name":"","num_papers_in_archive":9},{"url":"/dataset/smac-def-armored-parallel","name":"Def_Armored_parallel","full_name":"SMAC+_Def_Armored_parallel_20","num_papers_in_archive":8},{"url":"/dataset/smac-def-outnumbered-parallel","name":"Def_Outnumbered_parallel","full_name":"SMAC+_Def_Outnumbered_parallel_20","num_papers_in_archive":8},{"url":"/dataset/smac-off-hard-parallel","name":"Off_Hard_parallel","full_name":"SMAC+_Off_Hard_parallel_20","num_papers_in_archive":8},{"url":"/dataset/smac-off-superhard-parallel","name":"Off_Superhard_parallel","full_name":"SMAC+_Off_Superhard_parallel_20","num_papers_in_archive":8},{"url":"/dataset/smac-off-complicated-parallel","name":"Off_Complicated_parallel","full_name":"SMAC+_Off_Complicated_parallel_20","num_papers_in_archive":7},{"url":"/dataset/smac-off-distant-parallel","name":"Off_Distant_parallel","full_name":"SMAC+_Off_Distant_parallel_20","num_papers_in_archive":7},{"url":"/dataset/smac-off-near-parallel","name":"Off_Near_parallel","full_name":"SMAC+_Off_Near_parallel_20","num_papers_in_archive":7},{"url":"/dataset/smac-off-complicated-sequential","name":"Off_Complicated_sequential","full_name":"","num_papers_in_archive":5},{"url":"/dataset/smac-off-distant-sequential","name":"Off_Distant_sequential","full_name":"","num_papers_in_archive":5},{"url":"/dataset/smac-off-hard-sequential","name":"Off_Hard_sequential","full_name":"","num_papers_in_archive":5},{"url":"/dataset/smac-off-near-sequential","name":"Off_Near_sequential","full_name":"","num_papers_in_archive":5},{"url":"/dataset/smac-off-superhard-sequential","name":"Off_Superhard_sequential","full_name":"","num_papers_in_archive":5}],"subtasks":[],"parent_tasks":[{"url":"/task/smac","name":"SMAC"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":58,"tagged_in_all":126,"items":[{"url":"/paper/multi-agent-actor-critic-for-mixed","title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","date":"2017-06-07","arxiv_id":"1706.02275","repositories_listed":86,"syntology":{"n":143,"n_ran":75,"n_unverified":68,"n_pointer_only":99}},{"url":"/paper/the-starcraft-multi-agent-challenge","title":"The StarCraft Multi-Agent Challenge","date":"2019-02-11","arxiv_id":"1902.04043","repositories_listed":23,"syntology":{"n":15,"n_ran":6,"n_unverified":9,"n_pointer_only":13}},{"url":"/paper/qmix-monotonic-value-function-factorisation","title":"QMIX: Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","date":"2018-03-30","arxiv_id":"1803.11485","repositories_listed":18,"syntology":{"n":11,"n_ran":9,"n_unverified":2,"n_pointer_only":6}},{"url":"/paper/value-decomposition-networks-for-cooperative","title":"Value-Decomposition Networks For Cooperative Multi-Agent Learning","date":"2017-06-16","arxiv_id":"1706.05296","repositories_listed":10,"syntology":null},{"url":"/paper/is-independent-learning-all-you-need-in-the","title":"Is Independent Learning All You Need in the StarCraft Multi-Agent Challenge?","date":"2020-11-18","arxiv_id":"2011.09533","repositories_listed":7,"syntology":null},{"url":"/paper/counterfactual-multi-agent-policy-gradients","title":"Counterfactual Multi-Agent Policy Gradients","date":"2017-05-24","arxiv_id":"1705.08926","repositories_listed":7,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/maven-multi-agent-variational-exploration","title":"MAVEN: Multi-Agent Variational Exploration","date":"2019-10-16","arxiv_id":"1910.07483","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/qtran-learning-to-factorize-with","title":"QTRAN: Learning to Factorize with Transformation for Cooperative Multi-Agent Reinforcement Learning","date":"2019-05-14","arxiv_id":"1905.05408","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/mlrmbo-a-modular-framework-for-model-based","title":"mlrMBO: A Modular Framework for Model-Based Optimization of Expensive Black-Box Functions","date":"2017-03-09","arxiv_id":"1703.03373","repositories_listed":4,"syntology":null},{"url":"/paper/hyperparameter-tricks-in-multi-agent","title":"Rethinking the Implementation Matters in Cooperative Multi-Agent Reinforcement Learning","date":"2021-02-06","arxiv_id":"2102.03479","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/dfanet-deep-feature-aggregation-for-real-time","title":"DFANet: Deep Feature Aggregation for Real-Time Semantic Segmentation","date":"2019-04-03","arxiv_id":"1904.02216","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/generalizable-agent-modeling-for-agent","title":"Generalizable Agent Modeling for Agent Collaboration-Competition Adaptation with Multi-Retrieval and Dynamic Generation","date":"2025-06-20","arxiv_id":"2506.16718","repositories_listed":1,"syntology":null},{"url":"/paper/curriculum-learning-with-counterfactual-group","title":"Curriculum Learning With Counterfactual Group Relative Policy Advantage For Multi-Agent Reinforcement Learning","date":"2025-06-09","arxiv_id":"2506.07548","repositories_listed":1,"syntology":null},{"url":"/paper/jaxrobotarium-training-and-deploying-multi","title":"JaxRobotarium: Training and Deploying Multi-Robot Policies in 10 Minutes","date":"2025-05-10","arxiv_id":"2505.06771","repositories_listed":1,"syntology":null},{"url":"/paper/learning-generalizable-skills-from-offline","title":"Learning Generalizable Skills from Offline Multi-Task Data for Multi-Agent Cooperation","date":"2025-03-27","arxiv_id":"2503.21200","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/vlms-play-starcraft-ii-a-benchmark-and","title":"AVA: Attentive VLM Agent for Mastering StarCraft II","date":"2025-03-07","arxiv_id":"2503.05383","repositories_listed":1,"syntology":null},{"url":"/paper/an-extended-benchmarking-of-multi-agent","title":"An Extended Benchmarking of Multi-Agent Reinforcement Learning Algorithms in Complex Fully Cooperative Tasks","date":"2025-02-07","arxiv_id":"2502.04773","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/dual-ensembled-multiagent-q-learning-with","title":"Dual Ensembled Multiagent Q-Learning with Hypernet Regularizer","date":"2025-02-04","arxiv_id":"2502.02018","repositories_listed":1,"syntology":null},{"url":"/paper/smac-hard-enabling-mixed-opponent-strategy","title":"SMAC-Hard: Enabling Mixed Opponent Strategy Script and Self-play on SMAC","date":"2024-12-23","arxiv_id":"2412.17707","repositories_listed":1,"syntology":null},{"url":"/paper/llm-pysc2-starcraft-ii-learning-environment","title":"LLM-PySC2: Starcraft II learning environment for Large Language Models","date":"2024-11-08","arxiv_id":"2411.05348","repositories_listed":1,"syntology":null},{"url":"/paper/a-new-approach-to-solving-smac-task","title":"A New Approach to Solving SMAC Task: Generating Decision Tree Code from Large Language Models","date":"2024-10-21","arxiv_id":"2410.16024","repositories_listed":1,"syntology":null},{"url":"/paper/choices-are-more-important-than-efforts-llm","title":"Choices are More Important than Efforts: LLM Enables Efficient Multi-Agent Exploration","date":"2024-10-03","arxiv_id":"2410.02511","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/qtypemix-enhancing-multi-agent-cooperative","title":"QTypeMix: Enhancing Multi-Agent Cooperative Strategies through Heterogeneous and Homogeneous Value Decomposition","date":"2024-08-12","arxiv_id":"2408.07098","repositories_listed":1,"syntology":null},{"url":"/paper/decentralized-transformers-with-centralized","title":"Decentralized Transformers with Centralized Aggregation are Sample-Efficient Multi-Agent World Models","date":"2024-06-22","arxiv_id":"2406.15836","repositories_listed":1,"syntology":null},{"url":"/paper/soft-qmix-integrating-maximum-entropy-for","title":"Soft-QMIX: Integrating Maximum Entropy For Monotonic Value Function Factorization","date":"2024-06-20","arxiv_id":"2406.13930","repositories_listed":1,"syntology":null},{"url":"/paper/individual-contributions-as-intrinsic","title":"Individual Contributions as Intrinsic Exploration Scaffolds for Multi-agent Reinforcement Learning","date":"2024-05-28","arxiv_id":"2405.18110","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-multi-agent-reinforcement-learning","title":"Efficient Multi-agent Reinforcement Learning by Planning","date":"2024-05-20","arxiv_id":"2405.11778","repositories_listed":1,"syntology":null},{"url":"/paper/pps-qmix-periodically-parameter-sharing-for","title":"PPS-QMIX: Periodically Parameter Sharing for Accelerating Convergence of Multi-Agent Reinforcement Learning","date":"2024-03-05","arxiv_id":"2403.02635","repositories_listed":1,"syntology":null},{"url":"/paper/fox-formation-aware-exploration-in-multi","title":"FoX: Formation-aware exploration in multi-agent reinforcement learning","date":"2023-08-22","arxiv_id":"2308.11272","repositories_listed":1,"syntology":null},{"url":"/paper/homopt-a-homotopy-based-hyperparameter","title":"HomOpt: A Homotopy-Based Hyperparameter Optimization Method","date":"2023-08-07","arxiv_id":"2308.03317","repositories_listed":1,"syntology":null}],"syntology_records":11,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}