{"url":"/task/multi-agent-reinforcement-learning","name":"Multi-agent Reinforcement Learning","slug":"multi-agent-reinforcement-learning","description_markdown":"The target of **Multi-agent Reinforcement Learning** is to solve complex problems by integrating multiple agents that focus on different sub-tasks. In general, there are two types of multi-agent systems: independent and cooperative systems.\n\n\n<span class=\"description-source\">Source: [Show, Describe and Conclude: On Exploiting the Structure Information of Chest X-Ray Reports ](https://arxiv.org/abs/2004.12274)</span>","categories":[{"name":"Methodology","url":"/area/methodology"},{"name":"Playing Games","url":"/area/playing-games"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":1718,"papers_with_code":522,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":9,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/multi-agent-reinforcement-learning-on","slug":"multi-agent-reinforcement-learning-on","dataset":"ParticleEnvs Cooperative Communication","dataset_url":null,"rows_in_archive":1,"metrics":["final agent reward"],"first_row_in_archive_order":{"model":"MATD3","paper_title":"Reducing Overestimation Bias in Multi-Agent Domains Using Double Centralized Critics","paper_url":"/paper/reducing-overestimation-bias-in-multi-agent","paper_date":"2019-10-03","arxiv_id":"1910.01465","code_links":[{"title":"JohannesAck/tf2multiagentrl","url":"https://github.com/JohannesAck/tf2multiagentrl"},{"title":"JohannesAck/MATD3implementation","url":"https://github.com/JohannesAck/MATD3implementation"},{"title":"fdcl-gwu/gym-rotor","url":"https://github.com/fdcl-gwu/gym-rotor"}],"syntology":{"n":13,"n_ran":3,"n_unverified":10,"n_pointer_only":0}}},{"leaderboard":"/sota/multi-agent-reinforcement-learning-on-smac","slug":"multi-agent-reinforcement-learning-on-smac","dataset":"SMAC-Exp","dataset_url":"/dataset/smac-plus","rows_in_archive":1,"metrics":["Median Win Rate"],"first_row_in_archive_order":{"model":"DRIMA","paper_title":"Neural Processes with Stochastic Attention: Paying more attention to the context dataset","paper_url":"/paper/neural-processes-with-stochastic-attention-1","paper_date":"2022-04-11","arxiv_id":"2204.05449","code_links":[{"title":"mingyukim87/npwsa","url":"https://github.com/mingyukim87/npwsa"}],"syntology":null}},{"leaderboard":"/sota/multi-agent-reinforcement-learning-on-uav","slug":"multi-agent-reinforcement-learning-on-uav","dataset":"UAV Logistics","dataset_url":null,"rows_in_archive":1,"metrics":["Average Reward"],"first_row_in_archive_order":{"model":"Fusion-Multi-Actor-Attention-Critic","paper_title":"Multiagent Reinforcement Learning Based on Fusion-Multiactor-Attention-Critic for Multiple-Unmanned-Aerial-Vehicle Navigation Control","paper_url":"/paper/multiagent-reinforcement-learning-based-on","paper_date":"2022-10-10","arxiv_id":null,"code_links":[{"title":"leehe228/LogisticsEnv","url":"https://github.com/leehe228/LogisticsEnv"}],"syntology":null}}],"datasets":[{"url":"/dataset/cityflow","name":"CityFlow","full_name":"","num_papers_in_archive":47},{"url":"/dataset/starcraft-ii-learning-environment","name":"StarCraft II Learning Environment","full_name":"StarCraft II Learning Environment","num_papers_in_archive":26},{"url":"/dataset/smac-plus","name":"SMAC-Exp","full_name":"StarCraft Multi-Agent Exploration Challenge","num_papers_in_archive":11},{"url":"/dataset/hanabi-learning-environment","name":"Hanabi Learning Environment","full_name":"","num_papers_in_archive":10},{"url":"/dataset/og-marl","name":"OG-MARL","full_name":"Off-the-Grid MARL Datasets","num_papers_in_archive":6},{"url":"/dataset/civrealm","name":"CivRealm","full_name":"","num_papers_in_archive":4},{"url":"/dataset/colosseumrl","name":"ColosseumRL","full_name":"","num_papers_in_archive":1},{"url":"/dataset/pursuitmw","name":"pursuitMW","full_name":"Multi-agent pursuit in matrix world","num_papers_in_archive":1},{"url":"/dataset/roomenv","name":"RoomEnv-v0","full_name":"The Room environment - v0","num_papers_in_archive":1}],"subtasks":[{"url":"/task/smac","name":"SMAC"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":522,"tagged_in_all":1718,"items":[{"url":"/paper/multi-agent-actor-critic-for-mixed","title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","date":"2017-06-07","arxiv_id":"1706.02275","repositories_listed":86,"syntology":{"n":143,"n_ran":75,"n_unverified":68,"n_pointer_only":99}},{"url":"/paper/the-starcraft-multi-agent-challenge","title":"The StarCraft Multi-Agent Challenge","date":"2019-02-11","arxiv_id":"1902.04043","repositories_listed":23,"syntology":{"n":15,"n_ran":6,"n_unverified":9,"n_pointer_only":13}},{"url":"/paper/the-surprising-effectiveness-of-mappo-in","title":"The Surprising Effectiveness of PPO in Cooperative, Multi-Agent Games","date":"2021-03-02","arxiv_id":"2103.01955","repositories_listed":19,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/qmix-monotonic-value-function-factorisation","title":"QMIX: Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","date":"2018-03-30","arxiv_id":"1803.11485","repositories_listed":18,"syntology":{"n":11,"n_ran":9,"n_unverified":2,"n_pointer_only":6}},{"url":"/paper/trust-region-policy-optimisation-in-multi","title":"Trust Region Policy Optimisation in Multi-Agent Reinforcement Learning","date":"2021-09-23","arxiv_id":"2109.11251","repositories_listed":11,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/value-decomposition-networks-for-cooperative","title":"Value-Decomposition Networks For Cooperative Multi-Agent Learning","date":"2017-06-16","arxiv_id":"1706.05296","repositories_listed":10,"syntology":null},{"url":"/paper/rlcard-a-toolkit-for-reinforcement-learning","title":"RLCard: A Toolkit for Reinforcement Learning in Card Games","date":"2019-10-10","arxiv_id":"1910.04376","repositories_listed":9,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/is-independent-learning-all-you-need-in-the","title":"Is Independent Learning All You Need in the StarCraft Multi-Agent Challenge?","date":"2020-11-18","arxiv_id":"2011.09533","repositories_listed":7,"syntology":null},{"url":"/paper/qplex-duplex-dueling-multi-agent-q-learning","title":"QPLEX: Duplex Dueling Multi-Agent Q-Learning","date":"2020-08-03","arxiv_id":"2008.01062","repositories_listed":6,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/learning-with-opponent-learning-awareness","title":"Learning with Opponent-Learning Awareness","date":"2017-09-13","arxiv_id":"1709.04326","repositories_listed":6,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/smarts-scalable-multi-agent-reinforcement","title":"SMARTS: Scalable Multi-Agent Reinforcement Learning Training School for Autonomous Driving","date":"2020-10-19","arxiv_id":"2010.09776","repositories_listed":5,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":5}},{"url":"/paper/fully-decentralized-multi-agent-reinforcement","title":"Fully Decentralized Multi-Agent Reinforcement Learning with Networked Agents","date":"2018-02-23","arxiv_id":"1802.08757","repositories_listed":5,"syntology":null},{"url":"/paper/stabilising-experience-replay-for-deep-multi","title":"Stabilising Experience Replay for Deep Multi-Agent Reinforcement Learning","date":"2017-02-28","arxiv_id":"1702.08887","repositories_listed":5,"syntology":null},{"url":"/paper/sigmarl-a-sample-efficient-and-generalizable","title":"SigmaRL: A Sample-Efficient and Generalizable Multi-Agent Reinforcement Learning Framework for Motion Planning","date":"2024-08-14","arxiv_id":"2408.07644","repositories_listed":4,"syntology":null},{"url":"/paper/decom-decomposed-policy-for-constrained","title":"DeCOM: Decomposed Policy for Constrained Cooperative Multi-Agent Reinforcement Learning","date":"2021-11-10","arxiv_id":"2111.05670","repositories_listed":4,"syntology":null},{"url":"/paper/multi-agent-constrained-policy-optimisation","title":"Multi-Agent Constrained Policy Optimisation","date":"2021-10-06","arxiv_id":"2110.02793","repositories_listed":4,"syntology":{"n":7,"n_ran":1,"n_unverified":6,"n_pointer_only":1}},{"url":"/paper/sample-factory-egocentric-3d-control-from","title":"Sample Factory: Egocentric 3D Control from Pixels at 100000 FPS with Asynchronous Reinforcement Learning","date":"2020-06-21","arxiv_id":"2006.11751","repositories_listed":4,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/weighted-qmix-expanding-monotonic-value","title":"Weighted QMIX: Expanding Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","date":"2020-06-18","arxiv_id":"2006.10800","repositories_listed":4,"syntology":null},{"url":"/paper/191202288","title":"Simplified Action Decoder for Deep Multi-Agent Reinforcement Learning","date":"2019-12-04","arxiv_id":"1912.02288","repositories_listed":4,"syntology":null},{"url":"/paper/maven-multi-agent-variational-exploration","title":"MAVEN: Multi-Agent Variational Exploration","date":"2019-10-16","arxiv_id":"1910.07483","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/learning-transferable-cooperative-behavior-in","title":"Learning Transferable Cooperative Behavior in Multi-Agent Teams","date":"2019-06-04","arxiv_id":"1906.01202","repositories_listed":4,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/qtran-learning-to-factorize-with","title":"QTRAN: Learning to Factorize with Transformation for Cooperative Multi-Agent Reinforcement Learning","date":"2019-05-14","arxiv_id":"1905.05408","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/colight-learning-network-level-cooperation","title":"CoLight: Learning Network-level Cooperation for Traffic Signal Control","date":"2019-05-11","arxiv_id":"1905.05717","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/a-multi-agent-reinforcement-learning-model-of","title":"A multi-agent reinforcement learning model of common-pool resource appropriation","date":"2017-07-20","arxiv_id":"1707.06600","repositories_listed":4,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-in","title":"Multi-agent Reinforcement Learning in Sequential Social Dilemmas","date":"2017-02-10","arxiv_id":"1702.03037","repositories_listed":4,"syntology":null},{"url":"/paper/jaxmarl-multi-agent-rl-environments-in-jax","title":"JaxMARL: Multi-Agent RL Environments and Algorithms in JAX","date":"2023-11-16","arxiv_id":"2311.10090","repositories_listed":3,"syntology":{"n":23,"n_ran":0,"n_unverified":23,"n_pointer_only":0}},{"url":"/paper/deep-reinforcement-learning-for-multi-agent-2","title":"Deep Reinforcement Learning for Multi-Agent Interaction","date":"2022-08-02","arxiv_id":"2208.01769","repositories_listed":3,"syntology":null},{"url":"/paper/the-shapley-value-in-machine-learning","title":"The Shapley Value in Machine Learning","date":"2022-02-11","arxiv_id":"2202.05594","repositories_listed":3,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/warpdrive-extremely-fast-end-to-end-deep","title":"WarpDrive: Extremely Fast End-to-End Deep Multi-Agent Reinforcement Learning on a GPU","date":"2021-08-31","arxiv_id":"2108.13976","repositories_listed":3,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/learning-to-fly-a-gym-environment-with","title":"Learning to Fly -- a Gym Environment with PyBullet Physics for Reinforcement Learning of Multi-agent Quadcopter Control","date":"2021-03-03","arxiv_id":"2103.02142","repositories_listed":3,"syntology":null}],"syntology_records":18,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}