{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/towards-a-standardised-performance-evaluation","title":"Towards a Standardised Performance Evaluation Protocol for Cooperative MARL","arxiv_id":"2209.10485","date":"2022-09-21","proceeding":null,"authors":["Rihab Gorsane","Omayma Mahjoub","Ruan de Kock","Roland Dubb","Siddarth Singh","Arnu Pretorius"],"abstract":"Multi-agent reinforcement learning (MARL) has emerged as a useful approach to solving decentralised decision-making problems at scale. Research in the field has been growing steadily with many breakthrough algorithms proposed in recent years. In this work, we take a closer look at this rapid development with a focus on evaluation methodologies employed across a large body of research in cooperative MARL. By conducting a detailed meta-analysis of prior work, spanning 75 papers accepted for publication from 2016 to 2022, we bring to light worrying trends that put into question the true rate of progress. We further consider these trends in a wider context and take inspiration from single-agent RL literature on similar issues with recommendations that remain applicable to MARL. Combining these recommendations, with novel insights from our analysis, we propose a standardised performance evaluation protocol for cooperative MARL. We argue that such a standard protocol, if widely adopted, would greatly improve the validity and credibility of future research, make replication and reproducibility easier, as well as improve the ability of the field to accurately gauge the rate of progress over time by being able to make sound comparisons across different works. Finally, we release our meta-analysis data publicly on our project website for future research on evaluation: https://sites.google.com/view/marl-standard-protocol","url_abs":"https://arxiv.org/abs/2209.10485v1","url_pdf":"https://arxiv.org/pdf/2209.10485v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"towards-a-standardised-performance-evaluation","repo_url":"https://github.com/instadeepai/marl-eval","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"jax","reach":null}],"tasks":[{"task_slug":"decision-making","task_name":"Decision Making"},{"task_slug":"multi-agent-reinforcement-learning","task_name":"Multi-agent Reinforcement Learning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2209.10485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.10485"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"deterministic:regex_extraction","url":"https://github.com/google-research/rliable","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/instadeepai/marl-eval","reach":null}],"summary":{"ran_draft_wrong":3,"unverified":1},"by_repo_kind":{"listed":{"samples":4,"ran":3,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"ef74477c720f3bec","entry":"check_comma_in_algo_names","repo":"instadeepai/marl-eval","repo_kind":"listed","path":"marl_eval/utils/data_processing_utils.py","file_url":"https://github.com/instadeepai/marl-eval/blob/HEAD/marl_eval/utils/data_processing_utils.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"ef74477c720f3bec"}},{"code_sha256_prefix":"eda68a9d12eb7e6c","entry":"data_process_pipeline","repo":"instadeepai/marl-eval","repo_kind":"listed","path":"marl_eval/utils/data_processing_utils.py","file_url":"https://github.com/instadeepai/marl-eval/blob/HEAD/marl_eval/utils/data_processing_utils.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"eda68a9d12eb7e6c"}},{"code_sha256_prefix":"7d32b5690ab52952","entry":"lower_case_dictionary_keys","repo":"instadeepai/marl-eval","repo_kind":"listed","path":"marl_eval/utils/data_processing_utils.py","file_url":"https://github.com/instadeepai/marl-eval/blob/HEAD/marl_eval/utils/data_processing_utils.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"7d32b5690ab52952"}},{"code_sha256_prefix":"5e501c3503d155e9","entry":"lower_case_inputs","repo":"instadeepai/marl-eval","repo_kind":"listed","path":"marl_eval/utils/data_processing_utils.py","file_url":"https://github.com/instadeepai/marl-eval/blob/HEAD/marl_eval/utils/data_processing_utils.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"5e501c3503d155e9"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}