{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multi-agent-reinforcement-learning/papers/ran/1","list_of":"/task/multi-agent-reinforcement-learning","task":"Multi-agent Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":135,"counts":{"archive_papers_tagged":1718,"with_a_code_link":522,"where_syntology_ran_a_sample":135,"not_listed_spam_title":0,"listed":1718,"listed_where_code_ran":135,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":119,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":119,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multi-agent-reinforcement-learning/papers/ran/1","prev":null,"next":"/task/multi-agent-reinforcement-learning/papers/ran/2","papers":[{"url":"/paper/spiral-self-play-on-zero-sum-games","slug":"spiral-self-play-on-zero-sum-games","title":"SPIRAL: Self-Play on Zero-Sum Games Incentivizes Reasoning via Multi-Agent Multi-Turn Reinforcement Learning","date":"2025-06-30","arxiv_id":"2506.24119","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spiral-self-play-on-zero-sum-games#ran","syntology_url":"https://syntology.ai/paper/2506.24119","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.24119"}},"official":{"repos":["spiral-rl/spiral"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chasing-moving-targets-with-online-self-play","slug":"chasing-moving-targets-with-online-self-play","title":"Chasing Moving Targets with Online Self-Play Reinforcement Learning for Safer Language Models","date":"2025-06-09","arxiv_id":"2506.07468","repositories_listed":1,"syntology":{"n":17,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":11,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/chasing-moving-targets-with-online-self-play#ran","syntology_url":"https://syntology.ai/paper/2506.07468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.07468"}},"official":{"repos":["mickelliu/selfplay-redteaming"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":11,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/enhancing-cooperative-multi-agent","slug":"enhancing-cooperative-multi-agent","title":"Enhancing Cooperative Multi-Agent Reinforcement Learning with State Modelling and Adversarial Exploration","date":"2025-05-08","arxiv_id":"2505.05262","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":3,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/enhancing-cooperative-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2505.05262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05262"}},"official":{"repos":["ddaedalus/smpe"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/socialjax-an-evaluation-suite-for-multi-agent","slug":"socialjax-an-evaluation-suite-for-multi-agent","title":"SocialJax: An Evaluation Suite for Multi-agent Reinforcement Learning in Sequential Social Dilemmas","date":"2025-03-18","arxiv_id":"2503.14576","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/socialjax-an-evaluation-suite-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2503.14576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.14576"}},"official":{"repos":["cooperativex/socialjax"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exponential-topology-enabled-scalable","slug":"exponential-topology-enabled-scalable","title":"Exponential Topology-enabled Scalable Communication in Multi-agent Reinforcement Learning","date":"2025-02-27","arxiv_id":"2502.19717","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":1,"n_ran_checked":1,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exponential-topology-enabled-scalable#ran","syntology_url":"https://syntology.ai/paper/2502.19717","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19717"}},"official":{"repos":["lxxxxr/expocomm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official","unlocated"]}}},{"url":"/paper/routerl-multi-agent-reinforcement-learning","slug":"routerl-multi-agent-reinforcement-learning","title":"RouteRL: Multi-agent reinforcement learning framework for urban route choice with autonomous vehicles","date":"2025-02-27","arxiv_id":"2502.20065","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/routerl-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2502.20065","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20065"}},"official":{"repos":["coexistence-project/routerl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-solve-the-min-max-mixed-shelves","slug":"learning-to-solve-the-min-max-mixed-shelves","title":"Learning to Solve the Min-Max Mixed-Shelves Picker-Routing Problem via Hierarchical and Parallel Decoding","date":"2025-02-14","arxiv_id":"2502.10233","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-solve-the-min-max-mixed-shelves#ran","syntology_url":"https://syntology.ai/paper/2502.10233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.10233"}},"official":{"repos":["ltluttmann/marl4msprp"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-extended-benchmarking-of-multi-agent","slug":"an-extended-benchmarking-of-multi-agent","title":"An Extended Benchmarking of Multi-Agent Reinforcement Learning Algorithms in Complex Fully Cooperative Tasks","date":"2025-02-07","arxiv_id":"2502.04773","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-extended-benchmarking-of-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2502.04773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.04773"}},"official":{"repos":["ailabdsunipi/pymarlzooplus"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-with-focal","slug":"multi-agent-reinforcement-learning-with-focal","title":"Multi-Agent Reinforcement Learning with Focal Diversity Optimization","date":"2025-02-06","arxiv_id":"2502.04492","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/multi-agent-reinforcement-learning-with-focal#ran","syntology_url":"https://syntology.ai/paper/2502.04492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.04492"}},"official":{"repos":["sftekin/rl-focal"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wolfpack-adversarial-attack-for-robust-multi","slug":"wolfpack-adversarial-attack-for-robust-multi","title":"Wolfpack Adversarial Attack for Robust Multi-Agent Reinforcement Learning","date":"2025-02-05","arxiv_id":"2502.02844","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/wolfpack-adversarial-attack-for-robust-multi#ran","syntology_url":"https://syntology.ai/paper/2502.02844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02844"}},"official":{"repos":["sunwoolee0504/wall"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-retrieval-augmented-generation","slug":"improving-retrieval-augmented-generation","title":"Improving Retrieval-Augmented Generation through Multi-Agent Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15228","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2501.15228","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15228"}},"official":{"repos":["chenyiqun/mmoa-rag"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/camp-collaborative-attention-model-with","slug":"camp-collaborative-attention-model-with","title":"CAMP: Collaborative Attention Model with Profiles for Vehicle Routing Problems","date":"2025-01-06","arxiv_id":"2501.02977","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/camp-collaborative-attention-model-with#ran","syntology_url":"https://syntology.ai/paper/2501.02977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.02977"}},"official":{"repos":["ai4co/camp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-for-24","slug":"multi-agent-reinforcement-learning-for-24","title":"Multi Agent Reinforcement Learning for Sequential Satellite Assignment Problems","date":"2024-12-20","arxiv_id":"2412.15573","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-24#ran","syntology_url":"https://syntology.ai/paper/2412.15573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15573"}},"official":{"repos":["Rainlabuw/rl-enabled-distributed-assignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hypermarl-adaptive-hypernetworks-for-multi","slug":"hypermarl-adaptive-hypernetworks-for-multi","title":"HyperMARL: Adaptive Hypernetworks for Multi-Agent RL","date":"2024-12-05","arxiv_id":"2412.04233","repositories_listed":1,"syntology":{"n":26,"n_ran":11,"n_constructed":1,"n_ran_checked":10,"n_instrument":1,"n_unverified":15,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 1 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 15 unverified","sample_list":"/paper/hypermarl-adaptive-hypernetworks-for-multi#ran","syntology_url":"https://syntology.ai/paper/2412.04233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.04233"}},"official":{"repos":["kaleabtessera/hypermarl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":1,"n_ran_no_instrument_failure":10,"n_unverified":15,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-cooperate-with-humans-using","slug":"learning-to-cooperate-with-humans-using","title":"Learning to Cooperate with Humans using Generative Agents","date":"2024-11-21","arxiv_id":"2411.13934","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-cooperate-with-humans-using#ran","syntology_url":"https://syntology.ai/paper/2411.13934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13934"}},"official":{"repos":["lych1233/gamma-human-ai-collaboration"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adasociety-an-adaptive-environment-with","slug":"adasociety-an-adaptive-environment-with","title":"AdaSociety: An Adaptive Environment with Social Structures for Multi-Agent Decision-Making","date":"2024-11-06","arxiv_id":"2411.03865","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adasociety-an-adaptive-environment-with#ran","syntology_url":"https://syntology.ai/paper/2411.03865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03865"}},"official":{"repos":["bigai-ai/adasociety"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/intersectionzoo-eco-driving-for-benchmarking","slug":"intersectionzoo-eco-driving-for-benchmarking","title":"IntersectionZoo: Eco-driving for Benchmarking Multi-Agent Contextual Reinforcement Learning","date":"2024-10-19","arxiv_id":"2410.15221","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intersectionzoo-eco-driving-for-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2410.15221","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15221"}},"official":{"repos":["mit-wu-lab/IntersectionZoo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kaleidoscope-learnable-masks-for","slug":"kaleidoscope-learnable-masks-for","title":"Kaleidoscope: Learnable Masks for Heterogeneous Multi-agent Reinforcement Learning","date":"2024-10-11","arxiv_id":"2410.08540","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":2,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"11 ran (of which 2 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/kaleidoscope-learnable-masks-for#ran","syntology_url":"https://syntology.ai/paper/2410.08540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08540"}},"official":{"repos":["lxxxxr/kaleidoscope"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":2,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/coevolving-with-the-other-you-fine-tuning-llm","slug":"coevolving-with-the-other-you-fine-tuning-llm","title":"Coevolving with the Other You: Fine-Tuning LLM with Sequential Cooperative Multi-Agent Reinforcement Learning","date":"2024-10-08","arxiv_id":"2410.06101","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coevolving-with-the-other-you-fine-tuning-llm#ran","syntology_url":"https://syntology.ai/paper/2410.06101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06101"}},"official":{"repos":["Harry67Hu/CORY"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-training-for-enhanced-multi-task","slug":"hybrid-training-for-enhanced-multi-task","title":"Hybrid Training for Enhanced Multi-task Generalization in Multi-agent Reinforcement Learning","date":"2024-08-24","arxiv_id":"2408.13567","repositories_listed":0,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hybrid-training-for-enhanced-multi-task#ran","syntology_url":"https://syntology.ai/paper/2408.13567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.13567"}},"official":null}},{"url":"/paper/hokoff-real-game-dataset-from-honor-of-kings-1","slug":"hokoff-real-game-dataset-from-honor-of-kings-1","title":"Hokoff: Real Game Dataset from Honor of Kings and its Offline Reinforcement Learning Benchmarks","date":"2024-08-20","arxiv_id":"2408.10556","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hokoff-real-game-dataset-from-honor-of-kings-1#ran","syntology_url":"https://syntology.ai/paper/2408.10556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.10556"}},"official":{"repos":["tencent-ailab/hokoff"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sustaindc-benchmarking-for-sustainable-data","slug":"sustaindc-benchmarking-for-sustainable-data","title":"SustainDC: Benchmarking for Sustainable Data Center Control","date":"2024-08-14","arxiv_id":"2408.07841","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sustaindc-benchmarking-for-sustainable-data#ran","syntology_url":"https://syntology.ai/paper/2408.07841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07841"}},"official":{"repos":["hewlettpackard/dc-rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hypothetical-minds-scaffolding-theory-of-mind","slug":"hypothetical-minds-scaffolding-theory-of-mind","title":"Hypothetical Minds: Scaffolding Theory of Mind for Multi-Agent Tasks with Large Language Models","date":"2024-07-09","arxiv_id":"2407.07086","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hypothetical-minds-scaffolding-theory-of-mind#ran","syntology_url":"https://syntology.ai/paper/2407.07086","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07086"}},"official":{"repos":["locross93/hypothetical-minds"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dispelling-the-mirage-of-progress-in-offline","slug":"dispelling-the-mirage-of-progress-in-offline","title":"Dispelling the Mirage of Progress in Offline MARL through Standardised Baselines and Evaluation","date":"2024-06-13","arxiv_id":"2406.09068","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dispelling-the-mirage-of-progress-in-offline#ran","syntology_url":"https://syntology.ai/paper/2406.09068","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09068"}},"official":{"repos":["instadeepai/og-marl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/eqmarl-entangled-quantum-multi-agent","slug":"eqmarl-entangled-quantum-multi-agent","title":"eQMARL: Entangled Quantum Multi-Agent Reinforcement Learning for Distributed Cooperation over Quantum Channels","date":"2024-05-24","arxiv_id":"2405.17486","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/eqmarl-entangled-quantum-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2405.17486","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17486"}},"official":{"repos":["news-vt/eqmarl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/controlling-behavioral-diversity-in-multi","slug":"controlling-behavioral-diversity-in-multi","title":"Controlling Behavioral Diversity in Multi-Agent Reinforcement Learning","date":"2024-05-23","arxiv_id":"2405.15054","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/controlling-behavioral-diversity-in-multi#ran","syntology_url":"https://syntology.ai/paper/2405.15054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15054"}},"official":{"repos":["proroklab/controllingbehavioraldiversity"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/randomized-exploration-in-cooperative-multi","slug":"randomized-exploration-in-cooperative-multi","title":"Randomized Exploration in Cooperative Multi-Agent Reinforcement Learning","date":"2024-04-16","arxiv_id":"2404.10728","repositories_listed":0,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/randomized-exploration-in-cooperative-multi#ran","syntology_url":"https://syntology.ai/paper/2404.10728","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10728"}},"official":null}},{"url":"/paper/n-agent-ad-hoc-teamwork","slug":"n-agent-ad-hoc-teamwork","title":"N-Agent Ad Hoc Teamwork","date":"2024-04-16","arxiv_id":"2404.10740","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":7,"n_instrument":6,"n_unverified":5,"n_honours":4,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 4 honoured, 0 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/n-agent-ad-hoc-teamwork#ran","syntology_url":"https://syntology.ai/paper/2404.10740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10740"}},"official":{"repos":["carolinewang01/naht"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-episodic-memory-utilization-of","slug":"efficient-episodic-memory-utilization-of","title":"Efficient Episodic Memory Utilization of Cooperative Multi-Agent Reinforcement Learning","date":"2024-03-02","arxiv_id":"2403.01112","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/efficient-episodic-memory-utilization-of#ran","syntology_url":"https://syntology.ai/paper/2403.01112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01112"}},"official":{"repos":["hyunghona/emu"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-iterative-combinatorial-auction","slug":"understanding-iterative-combinatorial-auction","title":"Understanding Iterative Combinatorial Auction Designs via Multi-Agent Reinforcement Learning","date":"2024-02-29","arxiv_id":"2402.19420","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-iterative-combinatorial-auction#ran","syntology_url":"https://syntology.ai/paper/2402.19420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19420"}},"official":{"repos":["newmanne/open_spiel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conservative-and-risk-aware-offline-multi","slug":"conservative-and-risk-aware-offline-multi","title":"Conservative and Risk-Aware Offline Multi-Agent Reinforcement Learning","date":"2024-02-13","arxiv_id":"2402.08421","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conservative-and-risk-aware-offline-multi#ran","syntology_url":"https://syntology.ai/paper/2402.08421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08421"}},"official":{"repos":["eslam211/conservative-and-distributional-marl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-generalizability-of-multi-agent","slug":"towards-generalizability-of-multi-agent","title":"Towards Generalizability of Multi-Agent Reinforcement Learning in Graphs with Recurrent Message Passing","date":"2024-02-07","arxiv_id":"2402.05027","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-generalizability-of-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2402.05027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05027"}},"official":{"repos":["jw3il/graph-marl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/settling-decentralized-multi-agent","slug":"settling-decentralized-multi-agent","title":"Settling Decentralized Multi-Agent Coordinated Exploration by Novelty Sharing","date":"2024-02-03","arxiv_id":"2402.02097","repositories_listed":1,"syntology":{"n":18,"n_ran":16,"n_constructed":12,"n_ran_checked":15,"n_instrument":1,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":12,"n_pointer_only":18,"phrase":"16 ran (of which 12 constructed an object rather than computing a result; 15 with no instrument failure: 3 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/settling-decentralized-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2402.02097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02097"}},"official":{"repos":["sigmabm/mace"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":12,"n_ran_no_instrument_failure":15,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/fully-independent-communication-in-multi","slug":"fully-independent-communication-in-multi","title":"Fully Independent Communication in Multi-Agent Reinforcement Learning","date":"2024-01-26","arxiv_id":"2401.15059","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fully-independent-communication-in-multi#ran","syntology_url":"https://syntology.ai/paper/2401.15059","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.15059"}},"official":{"repos":["rafaelmp2/marl-indep-comm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarl-benchmarking-multi-agent","slug":"benchmarl-benchmarking-multi-agent","title":"BenchMARL: Benchmarking Multi-Agent Reinforcement Learning","date":"2023-12-03","arxiv_id":"2312.01472","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarl-benchmarking-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2312.01472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01472"}},"official":{"repos":["facebookresearch/benchmarl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/jaxmarl-multi-agent-rl-environments-in-jax","slug":"jaxmarl-multi-agent-rl-environments-in-jax","title":"JaxMARL: Multi-Agent RL Environments and Algorithms in JAX","date":"2023-11-16","arxiv_id":"2311.10090","repositories_listed":3,"syntology":{"n":23,"n_ran":22,"n_constructed":0,"n_ran_checked":22,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":22,"n_pointer_only":0,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 22 with no instrument failure: 0 honoured, 0 violated, 22 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/jaxmarl-multi-agent-rl-environments-in-jax#ran","syntology_url":"https://syntology.ai/paper/2311.10090","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10090"}},"official":{"repos":["flairox/jaxmarl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/multi-agent-quantum-reinforcement-learning","slug":"multi-agent-quantum-reinforcement-learning","title":"Multi-Agent Quantum Reinforcement Learning using Evolutionary Optimization","date":"2023-11-09","arxiv_id":"2311.05546","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-quantum-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2311.05546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05546"}},"official":{"repos":["michaelkoelle/qmarl-evo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/selectively-sharing-experiences-improves","slug":"selectively-sharing-experiences-improves","title":"Selectively Sharing Experiences Improves Multi-Agent Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00865","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/selectively-sharing-experiences-improves#ran","syntology_url":"https://syntology.ai/paper/2311.00865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.00865"}},"official":{"repos":["mgerstgrasser/super"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quantifying-zero-shot-coordination-capability","slug":"quantifying-zero-shot-coordination-capability","title":"ZSC-Eval: An Evaluation Toolkit and Benchmark for Multi-agent Zero-shot Coordination","date":"2023-10-08","arxiv_id":"2310.05208","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quantifying-zero-shot-coordination-capability#ran","syntology_url":"https://syntology.ai/paper/2310.05208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05208"}},"official":{"repos":["HumanCompatibleAI/overcooked_ai","sjtu-marl/zsc-eval"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-neuron-segmentation-with","slug":"self-supervised-neuron-segmentation-with","title":"Self-Supervised Neuron Segmentation with Multi-Agent Reinforcement Learning","date":"2023-10-06","arxiv_id":"2310.04148","repositories_listed":1,"syntology":{"n":24,"n_ran":20,"n_constructed":4,"n_ran_checked":18,"n_instrument":2,"n_unverified":4,"n_honours":2,"n_violates":1,"n_no_contract":15,"n_pointer_only":24,"phrase":"20 ran (of which 4 constructed an object rather than computing a result; 18 with no instrument failure: 2 honoured, 1 violated, 15 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/self-supervised-neuron-segmentation-with#ran","syntology_url":"https://syntology.ai/paper/2310.04148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04148"}},"official":{"repos":["ydchen0806/dbmim"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":4,"n_ran_no_instrument_failure":18,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/counterfactual-conservative-q-learning-for-1","slug":"counterfactual-conservative-q-learning-for-1","title":"Counterfactual Conservative Q Learning for Offline Multi-agent Reinforcement Learning","date":"2023-09-22","arxiv_id":"2309.12696","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":3,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":9,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/counterfactual-conservative-q-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2309.12696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12696"}},"official":{"repos":["thu-rllab/CFCQL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/towards-few-shot-coordination-revisiting-ad","slug":"towards-few-shot-coordination-revisiting-ad","title":"Towards Few-shot Coordination: Revisiting Ad-hoc Teamplay Challenge In the Game of Hanabi","date":"2023-08-20","arxiv_id":"2308.10284","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-few-shot-coordination-revisiting-ad#ran","syntology_url":"https://syntology.ai/paper/2308.10284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10284"}},"official":{"repos":["chandar-lab/adaptive-hanabi"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-multi-agent-reinforcement-learning-3","slug":"robust-multi-agent-reinforcement-learning-3","title":"Robust Multi-Agent Reinforcement Learning with State Uncertainty","date":"2023-07-30","arxiv_id":"2307.16212","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/robust-multi-agent-reinforcement-learning-3#ran","syntology_url":"https://syntology.ai/paper/2307.16212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16212"}},"official":{"repos":["sihongho/robust_marl_with_state_uncertainty"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-multi-agent-reinforcement-learning-1","slug":"offline-multi-agent-reinforcement-learning-1","title":"Offline Multi-Agent Reinforcement Learning with Implicit Global-to-Local Value Regularization","date":"2023-07-21","arxiv_id":"2307.11620","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/offline-multi-agent-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2307.11620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11620"}},"official":{"repos":["zhengyinan-air/omiga"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mediated-multi-agent-reinforcement-learning","slug":"mediated-multi-agent-reinforcement-learning","title":"Mediated Multi-Agent Reinforcement Learning","date":"2023-06-14","arxiv_id":"2306.08419","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mediated-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2306.08419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08419"}},"official":{"repos":["dimonenka/mediatedmarl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-versatile-multi-agent-reinforcement","slug":"a-versatile-multi-agent-reinforcement","title":"A Versatile Multi-Agent Reinforcement Learning Benchmark for Inventory Management","date":"2023-06-13","arxiv_id":"2306.07542","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-versatile-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.07542","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07542"}},"official":{"repos":["victoryxl/replenishmentenv"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ma2cl-masked-attentive-contrastive-learning","slug":"ma2cl-masked-attentive-contrastive-learning","title":"MA2CL:Masked Attentive Contrastive Learning for Multi-Agent Reinforcement Learning","date":"2023-06-03","arxiv_id":"2306.02006","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":4,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ma2cl-masked-attentive-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2306.02006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02006"}},"official":{"repos":["ustchlsong/ma2cl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/is-centralized-training-with-decentralized","slug":"is-centralized-training-with-decentralized","title":"Is Centralized Training with Decentralized Execution Framework Centralized Enough for MARL?","date":"2023-05-27","arxiv_id":"2305.17352","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/is-centralized-training-with-decentralized#ran","syntology_url":"https://syntology.ai/paper/2305.17352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17352"}},"official":{"repos":["zyh1999/cadp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-multi-agent-coordination-via","slug":"robust-multi-agent-coordination-via","title":"Robust multi-agent coordination via evolutionary generation of auxiliary adversarial attackers","date":"2023-05-10","arxiv_id":"2305.05909","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/robust-multi-agent-coordination-via#ran","syntology_url":"https://syntology.ai/paper/2305.05909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.05909"}},"official":{"repos":["zzq-bot/romance"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/local-optimization-achieves-global-optimality","slug":"local-optimization-achieves-global-optimality","title":"Local Optimization Achieves Global Optimality in Multi-Agent Reinforcement Learning","date":"2023-05-08","arxiv_id":"2305.04819","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/local-optimization-achieves-global-optimality#ran","syntology_url":"https://syntology.ai/paper/2305.04819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04819"}},"official":{"repos":["zhaoyl18/ratio_game"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/effective-and-stable-role-based-multi-agent","slug":"effective-and-stable-role-based-multi-agent","title":"Effective and Stable Role-Based Multi-Agent Collaboration by Structural Information Principles","date":"2023-04-03","arxiv_id":"2304.00755","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":4,"n_ran_checked":5,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":11,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/effective-and-stable-role-based-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2304.00755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.00755"}},"official":{"repos":["ringbdstack/sr-marl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mahtm-a-multi-agent-framework-for","slug":"mahtm-a-multi-agent-framework-for","title":"MAHTM: A Multi-Agent Framework for Hierarchical Transactive Microgrids","date":"2023-03-15","arxiv_id":"2303.08447","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/mahtm-a-multi-agent-framework-for#ran","syntology_url":"https://syntology.ai/paper/2303.08447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08447"}},"official":{"repos":["nicosquare/rl-energy-management"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/poseexaminer-automated-testing-of-out-of","slug":"poseexaminer-automated-testing-of-out-of","title":"PoseExaminer: Automated Testing of Out-of-Distribution Robustness in Human Pose and Shape Estimation","date":"2023-03-13","arxiv_id":"2303.07337","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/poseexaminer-automated-testing-of-out-of#ran","syntology_url":"https://syntology.ai/paper/2303.07337","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07337"}},"official":{"repos":["qihao067/poseexaminer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-zero-shot-cooperation-with-humans","slug":"learning-zero-shot-cooperation-with-humans","title":"Learning Zero-Shot Cooperation with Humans, Assuming Humans Are Biased","date":"2023-02-03","arxiv_id":"2302.01605","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-zero-shot-cooperation-with-humans#ran","syntology_url":"https://syntology.ai/paper/2302.01605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.01605"}},"official":{"repos":["samjia2000/HSP"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/off-the-grid-marl-a-framework-for-dataset","slug":"off-the-grid-marl-a-framework-for-dataset","title":"Off-the-Grid MARL: Datasets with Baselines for Offline Multi-Agent Reinforcement Learning","date":"2023-02-01","arxiv_id":"2302.00521","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/off-the-grid-marl-a-framework-for-dataset#ran","syntology_url":"https://syntology.ai/paper/2302.00521","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00521"}},"official":{"repos":["instadeepai/og-marl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-moral-choices-in-social-dilemmas","slug":"modeling-moral-choices-in-social-dilemmas","title":"Modeling Moral Choices in Social Dilemmas with Multi-Agent Reinforcement Learning","date":"2023-01-20","arxiv_id":"2301.08491","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modeling-moral-choices-in-social-dilemmas#ran","syntology_url":"https://syntology.ai/paper/2301.08491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.08491"}},"official":{"repos":["liza-tennant/moral_choice_dyadic","Liza-Tennant/modeling_moral_choice_dyadic"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/heterogeneous-multi-robot-reinforcement","slug":"heterogeneous-multi-robot-reinforcement","title":"Heterogeneous Multi-Robot Reinforcement Learning","date":"2023-01-17","arxiv_id":"2301.07137","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/heterogeneous-multi-robot-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2301.07137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.07137"}},"official":{"repos":["proroklab/hetgppo","proroklab/vectorizedmultiagentsimulator"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-multi-agent-reinforcement-learning-3","slug":"scalable-multi-agent-reinforcement-learning-3","title":"Scalable Multi-Agent Reinforcement Learning for Warehouse Logistics with Robotic and Human Co-Workers","date":"2022-12-22","arxiv_id":"2212.11498","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-multi-agent-reinforcement-learning-3#ran","syntology_url":"https://syntology.ai/paper/2212.11498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.11498"}},"official":{"repos":["uoe-agents/task-assignment-robotic-warehouse"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/effects-of-spectral-normalization-in-multi","slug":"effects-of-spectral-normalization-in-multi","title":"Effects of Spectral Normalization in Multi-agent Reinforcement Learning","date":"2022-12-10","arxiv_id":"2212.05331","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/effects-of-spectral-normalization-in-multi#ran","syntology_url":"https://syntology.ai/paper/2212.05331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05331"}},"official":{"repos":["kinalmehta/epymarl_spectral"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-is-the-solution-for-state-adversarial","slug":"what-is-the-solution-for-state-adversarial","title":"What is the Solution for State-Adversarial Multi-Agent Reinforcement Learning?","date":"2022-12-06","arxiv_id":"2212.02705","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/what-is-the-solution-for-state-adversarial#ran","syntology_url":"https://syntology.ai/paper/2212.02705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02705"}},"official":{"repos":["susanbao/rmarl_code"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/ace-cooperative-multi-agent-q-learning-with","slug":"ace-cooperative-multi-agent-q-learning-with","title":"ACE: Cooperative Multi-agent Q-learning with Bidirectional Action-Dependency","date":"2022-11-29","arxiv_id":"2211.16068","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ace-cooperative-multi-agent-q-learning-with#ran","syntology_url":"https://syntology.ai/paper/2211.16068","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.16068"}},"official":{"repos":["opendilab/ace"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-multi-agent-reinforcement-learning-2","slug":"scalable-multi-agent-reinforcement-learning-2","title":"Scalable Multi-Agent Reinforcement Learning through Intelligent Information Aggregation","date":"2022-11-03","arxiv_id":"2211.02127","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-multi-agent-reinforcement-learning-2#ran","syntology_url":"https://syntology.ai/paper/2211.02127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02127"}},"official":{"repos":["nsidn98/informarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-for-13","slug":"multi-agent-reinforcement-learning-for-13","title":"Multi-Agent Reinforcement Learning for Adaptive Mesh Refinement","date":"2022-11-02","arxiv_id":"2211.00801","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-13#ran","syntology_url":"https://syntology.ai/paper/2211.00801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00801"}},"official":{"repos":["011235813/marl-amr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-continuous-control-via-q-learning","slug":"solving-continuous-control-via-q-learning","title":"Solving Continuous Control via Q-learning","date":"2022-10-22","arxiv_id":"2210.12566","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/solving-continuous-control-via-q-learning#ran","syntology_url":"https://syntology.ai/paper/2210.12566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.12566"}},"official":{"repos":["tseyde/decqn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/proximal-learning-with-opponent-learning","slug":"proximal-learning-with-opponent-learning","title":"Proximal Learning With Opponent-Learning Awareness","date":"2022-10-18","arxiv_id":"2210.10125","repositories_listed":1,"syntology":{"n":20,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":13,"n_honours":5,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 5 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/proximal-learning-with-opponent-learning#ran","syntology_url":"https://syntology.ai/paper/2210.10125","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10125"}},"official":{"repos":["silent-zebra/pola"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":12,"ran_from_kinds":["official"]}}},{"url":"/paper/phantom-an-rl-driven-framework-for-agent","slug":"phantom-an-rl-driven-framework-for-agent","title":"Phantom -- A RL-driven multi-agent framework to model complex systems","date":"2022-10-12","arxiv_id":"2210.06012","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/phantom-an-rl-driven-framework-for-agent#ran","syntology_url":"https://syntology.ai/paper/2210.06012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06012"}},"official":{"repos":["jpmorganchase/Phantom"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/marllib-extending-rllib-for-multi-agent","slug":"marllib-extending-rllib-for-multi-agent","title":"MARLlib: A Scalable and Efficient Multi-agent Reinforcement Learning Library","date":"2022-10-11","arxiv_id":"2210.13708","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/marllib-extending-rllib-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2210.13708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13708"}},"official":{"repos":["replicable-marl/marllib"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stateful-active-facilitator-coordination-and","slug":"stateful-active-facilitator-coordination-and","title":"Stateful active facilitator: Coordination and Environmental Heterogeneity in Cooperative Multi-Agent Reinforcement Learning","date":"2022-10-04","arxiv_id":"2210.03022","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stateful-active-facilitator-coordination-and#ran","syntology_url":"https://syntology.ai/paper/2210.03022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03022"}},"official":{"repos":["jaggbow/saf","veds12/hecogrid"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-laws-for-a-multi-agent-reinforcement","slug":"scaling-laws-for-a-multi-agent-reinforcement","title":"Scaling Laws for a Multi-Agent Reinforcement Learning Model","date":"2022-09-29","arxiv_id":"2210.00849","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-laws-for-a-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.00849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00849"}},"official":{"repos":["orenneumann/alphazero-scaling-laws"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-a-standardised-performance-evaluation","slug":"towards-a-standardised-performance-evaluation","title":"Towards a Standardised Performance Evaluation Protocol for Cooperative MARL","date":"2022-09-21","arxiv_id":"2209.10485","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-a-standardised-performance-evaluation#ran","syntology_url":"https://syntology.ai/paper/2209.10485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.10485"}},"official":null}},{"url":"/paper/learning-sparse-graphon-mean-field-games","slug":"learning-sparse-graphon-mean-field-games","title":"Learning Sparse Graphon Mean Field Games","date":"2022-09-08","arxiv_id":"2209.03880","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-sparse-graphon-mean-field-games#ran","syntology_url":"https://syntology.ai/paper/2209.03880","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.03880"}},"official":{"repos":["chrfabian/learning_sparse_gmfgs"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/get-it-in-writing-formal-contracts-mitigate","slug":"get-it-in-writing-formal-contracts-mitigate","title":"Formal Contracts Mitigate Social Dilemmas in Multi-Agent RL","date":"2022-08-22","arxiv_id":"2208.10469","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/get-it-in-writing-formal-contracts-mitigate#ran","syntology_url":"https://syntology.ai/paper/2208.10469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.10469"}},"official":{"repos":["algorithmic-alignment-lab/contracts"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/last-iterate-convergence-with-full-and-noisy","slug":"last-iterate-convergence-with-full-and-noisy","title":"Last-Iterate Convergence with Full and Noisy Feedback in Two-Player Zero-Sum Games","date":"2022-08-21","arxiv_id":"2208.09855","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/last-iterate-convergence-with-full-and-noisy#ran","syntology_url":"https://syntology.ai/paper/2208.09855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.09855"}},"official":{"repos":["cyberagentailab/m2wu"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/heterogeneous-multi-agent-zero-shot","slug":"heterogeneous-multi-agent-zero-shot","title":"Heterogeneous Multi-agent Zero-Shot Coordination by Coevolution","date":"2022-08-09","arxiv_id":"2208.04957","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/heterogeneous-multi-agent-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2208.04957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.04957"}},"official":{"repos":["lamda-bbo/maze"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/pac-assisted-value-factorisation-with","slug":"pac-assisted-value-factorisation-with","title":"PAC: Assisted Value Factorisation with Counterfactual Predictions in Multi-Agent Reinforcement Learning","date":"2022-06-22","arxiv_id":"2206.11420","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/pac-assisted-value-factorisation-with#ran","syntology_url":"https://syntology.ai/paper/2206.11420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11420"}},"official":{"repos":["hanhananderson/pac-marl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-distributed-and-fair-policies-for","slug":"learning-distributed-and-fair-policies-for","title":"Learning Distributed and Fair Policies for Network Load Balancing as Markov Potential Game","date":"2022-06-03","arxiv_id":"2206.01451","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-distributed-and-fair-policies-for#ran","syntology_url":"https://syntology.ai/paper/2206.01451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01451"}},"official":{"repos":["zhiyuanyaoj/marllb"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/alma-hierarchical-learning-for-composite","slug":"alma-hierarchical-learning-for-composite","title":"ALMA: Hierarchical Learning for Composite Multi-Agent Tasks","date":"2022-05-27","arxiv_id":"2205.14205","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/alma-hierarchical-learning-for-composite#ran","syntology_url":"https://syntology.ai/paper/2205.14205","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14205"}},"official":{"repos":["shariqiqbal2810/alma"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pmic-improving-multi-agent-reinforcement-1","slug":"pmic-improving-multi-agent-reinforcement-1","title":"PMIC: Improving Multi-Agent Reinforcement Learning with Progressive Mutual Information Collaboration","date":"2022-03-16","arxiv_id":"2203.08553","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pmic-improving-multi-agent-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2203.08553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08553"}},"official":{"repos":["yeshenpy/pmic"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-shapley-value-in-machine-learning","slug":"the-shapley-value-in-machine-learning","title":"The Shapley Value in Machine Learning","date":"2022-02-11","arxiv_id":"2202.05594","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-shapley-value-in-machine-learning#ran","syntology_url":"https://syntology.ai/paper/2202.05594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.05594"}},"official":{"repos":["benedekrozemberczki/shapley"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/iterated-reasoning-with-mutual-information-in-1","slug":"iterated-reasoning-with-mutual-information-in-1","title":"Iterated Reasoning with Mutual Information in Cooperative and Byzantine Decentralized Teaming","date":"2022-01-20","arxiv_id":"2201.08484","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":1,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/iterated-reasoning-with-mutual-information-in-1#ran","syntology_url":"https://syntology.ai/paper/2201.08484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.08484"}},"official":{"repos":["core-robotics-lab/infopg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/multi-lingual-agents-through-multi-headed","slug":"multi-lingual-agents-through-multi-headed","title":"Multi-lingual agents through multi-headed neural networks","date":"2021-11-22","arxiv_id":"2111.11129","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-lingual-agents-through-multi-headed#ran","syntology_url":"https://syntology.ai/paper/2111.11129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.11129"}},"official":{"repos":["jon17591/multi-lingual-agents"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/off-policy-correction-for-multi-agent","slug":"off-policy-correction-for-multi-agent","title":"Off-Policy Correction For Multi-Agent Reinforcement Learning","date":"2021-11-22","arxiv_id":"2111.11229","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/off-policy-correction-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2111.11229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.11229"}},"official":{"repos":["awarelab/seed_rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/powergridworld-a-framework-for-multi-agent","slug":"powergridworld-a-framework-for-multi-agent","title":"PowerGridworld: A Framework for Multi-Agent Reinforcement Learning in Power Systems","date":"2021-11-10","arxiv_id":"2111.05969","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/powergridworld-a-framework-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2111.05969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.05969"}},"official":{"repos":["nrel/powergridworld"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/variational-automatic-curriculum-learning-for","slug":"variational-automatic-curriculum-learning-for","title":"Variational Automatic Curriculum Learning for Sparse-Reward Cooperative Multi-Agent Problems","date":"2021-11-08","arxiv_id":"2111.04613","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/variational-automatic-curriculum-learning-for#ran","syntology_url":"https://syntology.ai/paper/2111.04613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.04613"}},"official":null}},{"url":"/paper/learning-to-simulate-self-driven-particles","slug":"learning-to-simulate-self-driven-particles","title":"Learning to Simulate Self-Driven Particles System with Coordinated Policy Optimization","date":"2021-10-26","arxiv_id":"2110.13827","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/learning-to-simulate-self-driven-particles#ran","syntology_url":"https://syntology.ai/paper/2110.13827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.13827"}},"official":{"repos":["decisionforce/CoPO"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-constrained-policy-optimisation","slug":"multi-agent-constrained-policy-optimisation","title":"Multi-Agent Constrained Policy Optimisation","date":"2021-10-06","arxiv_id":"2110.02793","repositories_listed":4,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multi-agent-constrained-policy-optimisation#ran","syntology_url":"https://syntology.ai/paper/2110.02793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.02793"}},"official":{"repos":["chauncygu/multi-agent-constrained-policy-optimisation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/trust-region-policy-optimisation-in-multi","slug":"trust-region-policy-optimisation-in-multi","title":"Trust Region Policy Optimisation in Multi-Agent Reinforcement Learning","date":"2021-09-23","arxiv_id":"2109.11251","repositories_listed":11,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/trust-region-policy-optimisation-in-multi#ran","syntology_url":"https://syntology.ai/paper/2109.11251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.11251"}},"official":{"repos":["cyanrain7/trust-region-policy-optimisation-in-multi-agent-reinforcement-learning","anonymous-iclr22/trust-region-in-multi-agent-reinforcement-learning"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/warpdrive-extremely-fast-end-to-end-deep","slug":"warpdrive-extremely-fast-end-to-end-deep","title":"WarpDrive: Extremely Fast End-to-End Deep Multi-Agent Reinforcement Learning on a GPU","date":"2021-08-31","arxiv_id":"2108.13976","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/warpdrive-extremely-fast-end-to-end-deep#ran","syntology_url":"https://syntology.ai/paper/2108.13976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.13976"}},"official":{"repos":["salesforce/warp-drive"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/implicit-communication-as-minimum-entropy","slug":"implicit-communication-as-minimum-entropy","title":"Communicating via Markov Decision Processes","date":"2021-07-17","arxiv_id":"2107.08295","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/implicit-communication-as-minimum-entropy#ran","syntology_url":"https://syntology.ai/paper/2107.08295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.08295"}},"official":{"repos":["schroederdewitt/meme"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/douzero-mastering-doudizhu-with-self-play","slug":"douzero-mastering-doudizhu-with-self-play","title":"DouZero: Mastering DouDizhu with Self-Play Deep Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06135","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/douzero-mastering-doudizhu-with-self-play#ran","syntology_url":"https://syntology.ai/paper/2106.06135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06135"}},"official":{"repos":["kwai/DouZero"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/a-new-formalism-method-and-open-issues-for","slug":"a-new-formalism-method-and-open-issues-for","title":"A New Formalism, Method and Open Issues for Zero-Shot Coordination","date":"2021-06-11","arxiv_id":"2106.06613","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-new-formalism-method-and-open-issues-for#ran","syntology_url":"https://syntology.ai/paper/2106.06613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06613"}},"official":{"repos":["johannestreutlein/op-tie-breaking"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/malib-a-parallel-framework-for-population","slug":"malib-a-parallel-framework-for-population","title":"MALib: A Parallel Framework for Population-based Multi-agent Reinforcement Learning","date":"2021-06-05","arxiv_id":"2106.07551","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/malib-a-parallel-framework-for-population#ran","syntology_url":"https://syntology.ai/paper/2106.07551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.07551"}},"official":{"repos":["sjtu-marl/malib"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/coach-player-multi-agent-reinforcement","slug":"coach-player-multi-agent-reinforcement","title":"Coach-Player Multi-Agent Reinforcement Learning for Dynamic Team Composition","date":"2021-05-18","arxiv_id":"2105.08692","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coach-player-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2105.08692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.08692"}},"official":{"repos":["cranial-xix/marl-copa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-surprising-effectiveness-of-mappo-in","slug":"the-surprising-effectiveness-of-mappo-in","title":"The Surprising Effectiveness of PPO in Cooperative, Multi-Agent Games","date":"2021-03-02","arxiv_id":"2103.01955","repositories_listed":19,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-surprising-effectiveness-of-mappo-in#ran","syntology_url":"https://syntology.ai/paper/2103.01955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.01955"}},"official":{"repos":["marlbenchmark/on-policy"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/dfac-framework-factorizing-the-value-function","slug":"dfac-framework-factorizing-the-value-function","title":"DFAC Framework: Factorizing the Value Function via Quantile Mixture for Multi-Agent Distributional Q-Learning","date":"2021-02-16","arxiv_id":"2102.07936","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/dfac-framework-factorizing-the-value-function#ran","syntology_url":"https://syntology.ai/paper/2102.07936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.07936"}},"official":{"repos":["j3soon/dfac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-multi-agent-reinforcement-learning","slug":"scaling-multi-agent-reinforcement-learning","title":"Scaling Multi-Agent Reinforcement Learning with Selective Parameter Sharing","date":"2021-02-15","arxiv_id":"2102.07475","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/scaling-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2102.07475","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.07475"}},"official":{"repos":["uoe-agents/seps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hyperparameter-tricks-in-multi-agent","slug":"hyperparameter-tricks-in-multi-agent","title":"Rethinking the Implementation Matters in Cooperative Multi-Agent Reinforcement Learning","date":"2021-02-06","arxiv_id":"2102.03479","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hyperparameter-tricks-in-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2102.03479","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.03479"}},"official":{"repos":["hijkzzz/pymarl2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/multi-agent-reinforcement-learning-with","slug":"multi-agent-reinforcement-learning-with","title":"Multi-Agent Reinforcement Learning with Temporal Logic Specifications","date":"2021-02-01","arxiv_id":"2102.00582","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2102.00582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.00582"}},"official":{"repos":["lrhammond/almanac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-common-payoff-games-with-approximate","slug":"solving-common-payoff-games-with-approximate","title":"Solving Common-Payoff Games with Approximate Policy Iteration","date":"2021-01-11","arxiv_id":"2101.04237","repositories_listed":2,"syntology":{"n":9,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/solving-common-payoff-games-with-approximate#ran","syntology_url":"https://syntology.ai/paper/2101.04237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.04237"}},"official":{"repos":["ssokota/capi","ssokota/tiny-hanabi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-reinforcement-learning-policies-for","slug":"scalable-reinforcement-learning-policies-for","title":"Scalable Reinforcement Learning Policies for Multi-Agent Control","date":"2020-11-16","arxiv_id":"2011.08055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-reinforcement-learning-policies-for#ran","syntology_url":"https://syntology.ai/paper/2011.08055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.08055"}},"official":{"repos":["christopher-hsu/scalableMARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"a4d6c3d97c5f4c1ccefbd6c1a26dc9e06597518a80ff17bf290ffbdfa3100eeb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}