{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/multi-type-mean-field-reinforcement-learning","title":"Multi Type Mean Field Reinforcement Learning","arxiv_id":"2002.02513","date":"2020-02-06","proceeding":null,"authors":["Sriram Ganapathi Subramanian","Pascal Poupart","Matthew E. Taylor","Nidhi Hegde"],"abstract":"Mean field theory provides an effective way of scaling multiagent reinforcement learning algorithms to environments with many agents that can be abstracted by a virtual mean agent. In this paper, we extend mean field multiagent algorithms to multiple types. The types enable the relaxation of a core assumption in mean field reinforcement learning, which is that all agents in the environment are playing almost similar strategies and have the same goal. We conduct experiments on three different testbeds for the field of many agent reinforcement learning, based on the standard MAgents framework. We consider two different kinds of mean field environments: a) Games where agents belong to predefined types that are known a priori and b) Games where the type of each agent is unknown and therefore must be learned based on observations. We introduce new algorithms for each type of game and demonstrate their superior performance over state of the art algorithms that assume that all agents belong to the same type and other baseline algorithms in the MAgent framework.","url_abs":"https://arxiv.org/abs/2002.02513v7","url_pdf":"https://arxiv.org/pdf/2002.02513v7.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"multi-type-mean-field-reinforcement-learning","repo_url":"https://github.com/BorealisAI/mtmfrl","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"tf","reach":null}],"tasks":[{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"type","task_name":"Vocal Bursts Type Prediction"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2002.02513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.02513"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/BorealisAI/mtmfrl","reach":null}],"summary":{"ran_fixture":1,"ran_draft_wrong":1,"unverified":2},"by_repo_kind":{"official":{"samples":3,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":4,"samples":[{"code_sha256_prefix":"d2aaab57476c2fc1","entry":"linear_decay","repo":null,"repo_kind":null,"path":null,"file_url":null,"link_basis":"identical_code_first_harvested_elsewhere","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"mcp_get_code":{"code_sha256":"d2aaab57476c2fc1"}},{"code_sha256_prefix":"faa21a23d915dcee","entry":"play","repo":"BorealisAI/mtmfrl","repo_kind":"official","path":"multibattle/mfrl/examples/battle_model/senario_battle.py","file_url":"https://github.com/BorealisAI/mtmfrl/blob/HEAD/multibattle/mfrl/examples/battle_model/senario_battle.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"faa21a23d915dcee"}},{"code_sha256_prefix":"3d8ebd4702e55dc5","entry":"battle","repo":"BorealisAI/mtmfrl","repo_kind":"official","path":"multibattle/mfrl/examples/battle_model/senario_battle.py","file_url":"https://github.com/BorealisAI/mtmfrl/blob/HEAD/multibattle/mfrl/examples/battle_model/senario_battle.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"3d8ebd4702e55dc5"}},{"code_sha256_prefix":"2fac5442c1ec3859","entry":"play2","repo":"BorealisAI/mtmfrl","repo_kind":"official","path":"multibattle/mfrl/examples/battle_model/senario_battle.py","file_url":"https://github.com/BorealisAI/mtmfrl/blob/HEAD/multibattle/mfrl/examples/battle_model/senario_battle.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"2fac5442c1ec3859"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}