{"url":"/task/smac","name":"SMAC","slug":"smac","description_markdown":"The StarCraft Multi-Agent Challenge (SMAC) is a benchmark that provides elements of partial observability, challenging dynamics, and high-dimensional observation spaces. SMAC is built using the StarCraft II game engine, creating a testbed for research in cooperative MARL where each game unit is an independent RL agent.","categories":[{"name":"Playing Games","url":"/area/playing-games"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":121,"papers_with_code":54,"benchmarks":11,"benchmark_tables_in_archive":11,"benchmark_tables_shown":11,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":2,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/smac-on-smac-6h-vs-8z-1","slug":"smac-on-smac-6h-vs-8z-1","dataset":"SMAC 6h_vs_8z","dataset_url":"/dataset/smac","rows_in_archive":14,"metrics":["Median Win Rate","Average Score"],"first_row_in_archive_order":{"model":"ACE","paper_title":"ACE: Cooperative Multi-agent Q-learning with Bidirectional Action-Dependency","paper_url":"/paper/ace-cooperative-multi-agent-q-learning-with","paper_date":"2022-11-29","arxiv_id":"2211.16068","code_links":[{"title":"opendilab/ace","url":"https://github.com/opendilab/ace"}],"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}}},{"leaderboard":"/sota/smac-on-smac-mmm2-1","slug":"smac-on-smac-mmm2-1","dataset":"SMAC MMM2","dataset_url":"/dataset/smac","rows_in_archive":14,"metrics":["Median Win Rate","Average Score"],"first_row_in_archive_order":{"model":"ACE","paper_title":"ACE: Cooperative Multi-agent Q-learning with Bidirectional Action-Dependency","paper_url":"/paper/ace-cooperative-multi-agent-q-learning-with","paper_date":"2022-11-29","arxiv_id":"2211.16068","code_links":[{"title":"opendilab/ace","url":"https://github.com/opendilab/ace"}],"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}}},{"leaderboard":"/sota/smac-on-smac-3s5z-vs-3s6z-1","slug":"smac-on-smac-3s5z-vs-3s6z-1","dataset":"SMAC 3s5z_vs_3s6z","dataset_url":"/dataset/smac","rows_in_archive":13,"metrics":["Median Win Rate","Average Score"],"first_row_in_archive_order":{"model":"ACE","paper_title":"ACE: Cooperative Multi-agent Q-learning with Bidirectional Action-Dependency","paper_url":"/paper/ace-cooperative-multi-agent-q-learning-with","paper_date":"2022-11-29","arxiv_id":"2211.16068","code_links":[{"title":"opendilab/ace","url":"https://github.com/opendilab/ace"}],"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}}},{"leaderboard":"/sota/smac-on-smac-corridor","slug":"smac-on-smac-corridor","dataset":"SMAC corridor","dataset_url":"/dataset/smac","rows_in_archive":13,"metrics":["Median Win Rate","Average Score"],"first_row_in_archive_order":{"model":"ACE","paper_title":"ACE: Cooperative Multi-agent Q-learning with Bidirectional Action-Dependency","paper_url":"/paper/ace-cooperative-multi-agent-q-learning-with","paper_date":"2022-11-29","arxiv_id":"2211.16068","code_links":[{"title":"opendilab/ace","url":"https://github.com/opendilab/ace"}],"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}}},{"leaderboard":"/sota/smac-on-smac-27m-vs-30m","slug":"smac-on-smac-27m-vs-30m","dataset":"SMAC 27m_vs_30m","dataset_url":"/dataset/smac","rows_in_archive":11,"metrics":["Median Win Rate","Average Score"],"first_row_in_archive_order":{"model":"DDN","paper_title":"DFAC Framework: Factorizing the Value Function via Quantile Mixture for Multi-Agent Distributional Q-Learning","paper_url":"/paper/dfac-framework-factorizing-the-value-function","paper_date":"2021-02-16","arxiv_id":"2102.07936","code_links":[{"title":"j3soon/dfac","url":"https://github.com/j3soon/dfac"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/smac-on-smac-26m-vs-30m","slug":"smac-on-smac-26m-vs-30m","dataset":"SMAC 26m_vs_30m","dataset_url":null,"rows_in_archive":6,"metrics":["Average Score","Median Win Rate"],"first_row_in_archive_order":{"model":"DMIX","paper_title":"A Unified Framework for Factorizing Distributional Value Functions for Multi-Agent Reinforcement Learning","paper_url":"/paper/a-unified-framework-for-factorizing","paper_date":"2023-06-04","arxiv_id":"2306.02430","code_links":[{"title":"j3soon/dfac-extended","url":"https://github.com/j3soon/dfac-extended"}],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-3s5z-vs-4s6z","slug":"smac-on-smac-3s5z-vs-4s6z","dataset":"SMAC 3s5z_vs_4s6z","dataset_url":null,"rows_in_archive":6,"metrics":["Average Score","Median Win Rate"],"first_row_in_archive_order":{"model":"DDN","paper_title":"A Unified Framework for Factorizing Distributional Value Functions for Multi-Agent Reinforcement Learning","paper_url":"/paper/a-unified-framework-for-factorizing","paper_date":"2023-06-04","arxiv_id":"2306.02430","code_links":[{"title":"j3soon/dfac-extended","url":"https://github.com/j3soon/dfac-extended"}],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-6h-vs-9z","slug":"smac-on-smac-6h-vs-9z","dataset":"SMAC 6h_vs_9z","dataset_url":null,"rows_in_archive":6,"metrics":["Average Score","Median Win Rate"],"first_row_in_archive_order":{"model":"DDN","paper_title":"A Unified Framework for Factorizing Distributional Value Functions for Multi-Agent Reinforcement Learning","paper_url":"/paper/a-unified-framework-for-factorizing","paper_date":"2023-06-04","arxiv_id":"2306.02430","code_links":[{"title":"j3soon/dfac-extended","url":"https://github.com/j3soon/dfac-extended"}],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-corridor-2z-vs-24zg","slug":"smac-on-smac-corridor-2z-vs-24zg","dataset":"SMAC corridor_2z_vs_24zg","dataset_url":null,"rows_in_archive":6,"metrics":["Average Score","Median Win Rate"],"first_row_in_archive_order":{"model":"DDN","paper_title":"A Unified Framework for Factorizing Distributional Value Functions for Multi-Agent Reinforcement Learning","paper_url":"/paper/a-unified-framework-for-factorizing","paper_date":"2023-06-04","arxiv_id":"2306.02430","code_links":[{"title":"j3soon/dfac-extended","url":"https://github.com/j3soon/dfac-extended"}],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-mmm2-7m2m1m-vs-8m4m1m","slug":"smac-on-smac-mmm2-7m2m1m-vs-8m4m1m","dataset":"SMAC MMM2_7m2M1M_vs_8m4M1M","dataset_url":null,"rows_in_archive":6,"metrics":["Average Score","Median Win Rate"],"first_row_in_archive_order":{"model":"DDN","paper_title":"A Unified Framework for Factorizing Distributional Value Functions for Multi-Agent Reinforcement Learning","paper_url":"/paper/a-unified-framework-for-factorizing","paper_date":"2023-06-04","arxiv_id":"2306.02430","code_links":[{"title":"j3soon/dfac-extended","url":"https://github.com/j3soon/dfac-extended"}],"syntology":null}},{"leaderboard":"/sota/smac-on-smac-mmm2-7m2m1m-vs-9m3m1m","slug":"smac-on-smac-mmm2-7m2m1m-vs-9m3m1m","dataset":"SMAC MMM2_7m2M1M_vs_9m3M1M","dataset_url":null,"rows_in_archive":6,"metrics":["Average Score","Median Win Rate"],"first_row_in_archive_order":{"model":"DDN","paper_title":"A Unified Framework for Factorizing Distributional Value Functions for Multi-Agent Reinforcement Learning","paper_url":"/paper/a-unified-framework-for-factorizing","paper_date":"2023-06-04","arxiv_id":"2306.02430","code_links":[{"title":"j3soon/dfac-extended","url":"https://github.com/j3soon/dfac-extended"}],"syntology":null}}],"datasets":[{"url":"/dataset/smac","name":"SMAC","full_name":"The StarCraft Multi-Agent Challenge","num_papers_in_archive":324}],"subtasks":[{"url":"/task/smac-1","name":"SMAC+"},{"url":"/task/smac-plus","name":"SMAC Plus"}],"parent_tasks":[{"url":"/task/multi-agent-reinforcement-learning","name":"Multi-agent Reinforcement Learning"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":54,"tagged_in_all":121,"items":[{"url":"/paper/the-starcraft-multi-agent-challenge","title":"The StarCraft Multi-Agent Challenge","date":"2019-02-11","arxiv_id":"1902.04043","repositories_listed":23,"syntology":{"n":15,"n_ran":6,"n_unverified":9,"n_pointer_only":13}},{"url":"/paper/is-independent-learning-all-you-need-in-the","title":"Is Independent Learning All You Need in the StarCraft Multi-Agent Challenge?","date":"2020-11-18","arxiv_id":"2011.09533","repositories_listed":7,"syntology":null},{"url":"/paper/maven-multi-agent-variational-exploration","title":"MAVEN: Multi-Agent Variational Exploration","date":"2019-10-16","arxiv_id":"1910.07483","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/mlrmbo-a-modular-framework-for-model-based","title":"mlrMBO: A Modular Framework for Model-Based Optimization of Expensive Black-Box Functions","date":"2017-03-09","arxiv_id":"1703.03373","repositories_listed":4,"syntology":null},{"url":"/paper/deep-multi-agent-reinforcement-learning-for","title":"FACMAC: Factored Multi-Agent Centralised Policy Gradients","date":"2020-03-14","arxiv_id":"2003.06709","repositories_listed":3,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/hyperparameter-tricks-in-multi-agent","title":"Rethinking the Implementation Matters in Cooperative Multi-Agent Reinforcement Learning","date":"2021-02-06","arxiv_id":"2102.03479","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/generalizable-agent-modeling-for-agent","title":"Generalizable Agent Modeling for Agent Collaboration-Competition Adaptation with Multi-Retrieval and Dynamic Generation","date":"2025-06-20","arxiv_id":"2506.16718","repositories_listed":1,"syntology":null},{"url":"/paper/curriculum-learning-with-counterfactual-group","title":"Curriculum Learning With Counterfactual Group Relative Policy Advantage For Multi-Agent Reinforcement Learning","date":"2025-06-09","arxiv_id":"2506.07548","repositories_listed":1,"syntology":null},{"url":"/paper/jaxrobotarium-training-and-deploying-multi","title":"JaxRobotarium: Training and Deploying Multi-Robot Policies in 10 Minutes","date":"2025-05-10","arxiv_id":"2505.06771","repositories_listed":1,"syntology":null},{"url":"/paper/learning-generalizable-skills-from-offline","title":"Learning Generalizable Skills from Offline Multi-Task Data for Multi-Agent Cooperation","date":"2025-03-27","arxiv_id":"2503.21200","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/vlms-play-starcraft-ii-a-benchmark-and","title":"AVA: Attentive VLM Agent for Mastering StarCraft II","date":"2025-03-07","arxiv_id":"2503.05383","repositories_listed":1,"syntology":null},{"url":"/paper/an-extended-benchmarking-of-multi-agent","title":"An Extended Benchmarking of Multi-Agent Reinforcement Learning Algorithms in Complex Fully Cooperative Tasks","date":"2025-02-07","arxiv_id":"2502.04773","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/dual-ensembled-multiagent-q-learning-with","title":"Dual Ensembled Multiagent Q-Learning with Hypernet Regularizer","date":"2025-02-04","arxiv_id":"2502.02018","repositories_listed":1,"syntology":null},{"url":"/paper/smac-hard-enabling-mixed-opponent-strategy","title":"SMAC-Hard: Enabling Mixed Opponent Strategy Script and Self-play on SMAC","date":"2024-12-23","arxiv_id":"2412.17707","repositories_listed":1,"syntology":null},{"url":"/paper/llm-pysc2-starcraft-ii-learning-environment","title":"LLM-PySC2: Starcraft II learning environment for Large Language Models","date":"2024-11-08","arxiv_id":"2411.05348","repositories_listed":1,"syntology":null},{"url":"/paper/a-new-approach-to-solving-smac-task","title":"A New Approach to Solving SMAC Task: Generating Decision Tree Code from Large Language Models","date":"2024-10-21","arxiv_id":"2410.16024","repositories_listed":1,"syntology":null},{"url":"/paper/choices-are-more-important-than-efforts-llm","title":"Choices are More Important than Efforts: LLM Enables Efficient Multi-Agent Exploration","date":"2024-10-03","arxiv_id":"2410.02511","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/qtypemix-enhancing-multi-agent-cooperative","title":"QTypeMix: Enhancing Multi-Agent Cooperative Strategies through Heterogeneous and Homogeneous Value Decomposition","date":"2024-08-12","arxiv_id":"2408.07098","repositories_listed":1,"syntology":null},{"url":"/paper/decentralized-transformers-with-centralized","title":"Decentralized Transformers with Centralized Aggregation are Sample-Efficient Multi-Agent World Models","date":"2024-06-22","arxiv_id":"2406.15836","repositories_listed":1,"syntology":null},{"url":"/paper/soft-qmix-integrating-maximum-entropy-for","title":"Soft-QMIX: Integrating Maximum Entropy For Monotonic Value Function Factorization","date":"2024-06-20","arxiv_id":"2406.13930","repositories_listed":1,"syntology":null},{"url":"/paper/individual-contributions-as-intrinsic","title":"Individual Contributions as Intrinsic Exploration Scaffolds for Multi-agent Reinforcement Learning","date":"2024-05-28","arxiv_id":"2405.18110","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-multi-agent-reinforcement-learning","title":"Efficient Multi-agent Reinforcement Learning by Planning","date":"2024-05-20","arxiv_id":"2405.11778","repositories_listed":1,"syntology":null},{"url":"/paper/pps-qmix-periodically-parameter-sharing-for","title":"PPS-QMIX: Periodically Parameter Sharing for Accelerating Convergence of Multi-Agent Reinforcement Learning","date":"2024-03-05","arxiv_id":"2403.02635","repositories_listed":1,"syntology":null},{"url":"/paper/fox-formation-aware-exploration-in-multi","title":"FoX: Formation-aware exploration in multi-agent reinforcement learning","date":"2023-08-22","arxiv_id":"2308.11272","repositories_listed":1,"syntology":null},{"url":"/paper/homopt-a-homotopy-based-hyperparameter","title":"HomOpt: A Homotopy-Based Hyperparameter Optimization Method","date":"2023-08-07","arxiv_id":"2308.03317","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-framework-for-factorizing","title":"A Unified Framework for Factorizing Distributional Value Functions for Multi-Agent Reinforcement Learning","date":"2023-06-04","arxiv_id":"2306.02430","repositories_listed":1,"syntology":null},{"url":"/paper/robust-multi-agent-coordination-via","title":"Robust multi-agent coordination via evolutionary generation of auxiliary adversarial attackers","date":"2023-05-10","arxiv_id":"2305.05909","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/smaclite-a-lightweight-environment-for-multi","title":"SMAClite: A Lightweight Environment for Multi-Agent Reinforcement Learning","date":"2023-05-09","arxiv_id":"2305.05566","repositories_listed":1,"syntology":null},{"url":"/paper/ghq-grouped-hybrid-q-learning-for","title":"GHQ: Grouped Hybrid Q Learning for Heterogeneous Cooperative Multi-agent Reinforcement Learning","date":"2023-03-02","arxiv_id":"2303.01070","repositories_listed":1,"syntology":null},{"url":"/paper/attacking-cooperative-multi-agent","title":"Attacking Cooperative Multi-Agent Reinforcement Learning by Adversarial Minority Influence","date":"2023-02-07","arxiv_id":"2302.03322","repositories_listed":1,"syntology":null}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}