{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/mean-field-multi-agent-reinforcement-learning","title":"Mean Field Multi-Agent Reinforcement Learning","arxiv_id":"1802.05438","date":"2018-02-15","proceeding":"ICML 2018 7","authors":["Yaodong Yang","Rui Luo","Minne Li","Ming Zhou","Wei-Nan Zhang","Jun Wang"],"abstract":"Existing multi-agent reinforcement learning methods are limited typically to a small number of agents. When the agent number increases largely, the learning becomes intractable due to the curse of the dimensionality and the exponential growth of agent interactions. In this paper, we present \\emph{Mean Field Reinforcement Learning} where the interactions within the population of agents are approximated by those between a single agent and the average effect from the overall population or neighboring agents; the interplay between the two entities is mutually reinforced: the learning of the individual agent's optimal policy depends on the dynamics of the population, while the dynamics of the population change according to the collective patterns of the individual policies. We develop practical mean field Q-learning and mean field Actor-Critic algorithms and analyze the convergence of the solution to Nash equilibrium. Experiments on Gaussian squeeze, Ising model, and battle games justify the learning effectiveness of our mean field approaches. In addition, we report the first result to solve the Ising model via model-free reinforcement learning methods.","url_abs":"https://arxiv.org/abs/1802.05438v5","url_pdf":"https://arxiv.org/pdf/1802.05438v5.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"mean-field-multi-agent-reinforcement-learning","repo_url":"https://github.com/mlii/mfrl","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"mean-field-multi-agent-reinforcement-learning","repo_url":"https://github.com/baoqianwang/iros22_darl1n","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"mean-field-multi-agent-reinforcement-learning","repo_url":"https://github.com/esmeralday/MARL","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}}],"tasks":[{"task_slug":"multi-agent-reinforcement-learning","task_name":"Multi-agent Reinforcement Learning"},{"task_slug":"q-learning","task_name":"Q-Learning"},{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[{"method_slug":"q-learning","method_name":"Q-Learning"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1802.05438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.05438"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/baoqianwang/iros22_darl1n","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mlii/mfrl","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/esmeralday/MARL","reach":{"status":"ok"}}],"summary":{"ran_fixture":2,"unverified":1},"by_repo_kind":{"official":{"samples":1,"ran":1,"repositories":1},"listed":{"samples":2,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":2,"samples":[{"code_sha256_prefix":"d2aaab57476c2fc1","entry":"linear_decay","repo":"mlii/mfrl","repo_kind":"official","path":"train_battle.py","file_url":"https://github.com/mlii/mfrl/blob/HEAD/train_battle.py","link_basis":"plan_row","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"d2aaab57476c2fc1"}},{"code_sha256_prefix":"e6e29c2c0e881c3c","entry":"make_env","repo":"baoqianwang/iros22_darl1n","repo_kind":"listed","path":"maddpg_o/experiments/train_normal.py","file_url":"https://github.com/baoqianwang/iros22_darl1n/blob/HEAD/maddpg_o/experiments/train_normal.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"DEP_MISSING","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"e6e29c2c0e881c3c"}},{"code_sha256_prefix":"b6a7ee165ede9526","entry":"make_env","repo":"baoqianwang/iros22_darl1n","repo_kind":"listed","path":"maddpg_o/experiments/train_darl1n.py","file_url":"https://github.com/baoqianwang/iros22_darl1n/blob/HEAD/maddpg_o/experiments/train_darl1n.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"b6a7ee165ede9526"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}