{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multi-agent-reinforcement-learning/papers/3","list_of":"/task/multi-agent-reinforcement-learning","task":"Multi-agent Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":18,"rows_per_page":100,"rows":[201,300],"of":1718,"counts":{"archive_papers_tagged":1718,"with_a_code_link":522,"where_syntology_ran_a_sample":135,"not_listed_spam_title":0,"listed":1718,"listed_where_code_ran":135,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":119,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":119,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multi-agent-reinforcement-learning","prev":"/task/multi-agent-reinforcement-learning/papers/2","next":"/task/multi-agent-reinforcement-learning/papers/4","papers":[{"url":"/paper/understanding-iterative-combinatorial-auction","slug":"understanding-iterative-combinatorial-auction","title":"Understanding Iterative Combinatorial Auction Designs via Multi-Agent Reinforcement Learning","date":"2024-02-29","arxiv_id":"2402.19420","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-iterative-combinatorial-auction#ran","syntology_url":"https://syntology.ai/paper/2402.19420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19420"}},"official":{"repos":["newmanne/open_spiel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/independent-learning-in-constrained-markov","slug":"independent-learning-in-constrained-markov","title":"Independent Learning in Constrained Markov Potential Games","date":"2024-02-27","arxiv_id":"2402.17885","repositories_listed":1,"syntology":null},{"url":"/paper/modelling-crypto-markets-by-multi-agent","slug":"modelling-crypto-markets-by-multi-agent","title":"Modelling crypto markets by multi-agent reinforcement learning","date":"2024-02-16","arxiv_id":"2402.10803","repositories_listed":1,"syntology":null},{"url":"/paper/conservative-and-risk-aware-offline-multi","slug":"conservative-and-risk-aware-offline-multi","title":"Conservative and Risk-Aware Offline Multi-Agent Reinforcement Learning","date":"2024-02-13","arxiv_id":"2402.08421","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conservative-and-risk-aware-offline-multi#ran","syntology_url":"https://syntology.ai/paper/2402.08421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08421"}},"official":{"repos":["eslam211/conservative-and-distributional-marl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/risk-sensitive-multi-agent-reinforcement","slug":"risk-sensitive-multi-agent-reinforcement","title":"Risk-Sensitive Multi-Agent Reinforcement Learning in Network Aggregative Markov Games","date":"2024-02-08","arxiv_id":"2402.05906","repositories_listed":1,"syntology":null},{"url":"/paper/towards-generalizability-of-multi-agent","slug":"towards-generalizability-of-multi-agent","title":"Towards Generalizability of Multi-Agent Reinforcement Learning in Graphs with Recurrent Message Passing","date":"2024-02-07","arxiv_id":"2402.05027","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-generalizability-of-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2402.05027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05027"}},"official":{"repos":["jw3il/graph-marl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sub-play-adversarial-policies-against","slug":"sub-play-adversarial-policies-against","title":"SUB-PLAY: Adversarial Policies against Partially Observed Multi-Agent Reinforcement Learning Systems","date":"2024-02-06","arxiv_id":"2402.03741","repositories_listed":1,"syntology":null},{"url":"/paper/settling-decentralized-multi-agent","slug":"settling-decentralized-multi-agent","title":"Settling Decentralized Multi-Agent Coordinated Exploration by Novelty Sharing","date":"2024-02-03","arxiv_id":"2402.02097","repositories_listed":1,"syntology":{"n":18,"n_ran":16,"n_constructed":12,"n_ran_checked":15,"n_instrument":1,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":12,"n_pointer_only":18,"phrase":"16 ran (of which 12 constructed an object rather than computing a result; 15 with no instrument failure: 3 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/settling-decentralized-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2402.02097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02097"}},"official":{"repos":["sigmabm/mace"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":12,"n_ran_no_instrument_failure":15,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/developing-a-multi-agent-and-self-adaptive","slug":"developing-a-multi-agent-and-self-adaptive","title":"Developing A Multi-Agent and Self-Adaptive Framework with Deep Reinforcement Learning for Dynamic Portfolio Risk Management","date":"2024-02-01","arxiv_id":"2402.00515","repositories_listed":1,"syntology":null},{"url":"/paper/fully-independent-communication-in-multi","slug":"fully-independent-communication-in-multi","title":"Fully Independent Communication in Multi-Agent Reinforcement Learning","date":"2024-01-26","arxiv_id":"2401.15059","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fully-independent-communication-in-multi#ran","syntology_url":"https://syntology.ai/paper/2401.15059","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.15059"}},"official":{"repos":["rafaelmp2/marl-indep-comm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/emergent-dominance-hierarchies-in","slug":"emergent-dominance-hierarchies-in","title":"Emergent Dominance Hierarchies in Reinforcement Learning Agents","date":"2024-01-21","arxiv_id":"2401.12258","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-policy-distance-for-multi-agent","slug":"measuring-policy-distance-for-multi-agent","title":"Measuring Policy Distance for Multi-Agent Reinforcement Learning","date":"2024-01-20","arxiv_id":"2401.11257","repositories_listed":1,"syntology":null},{"url":"/paper/aquarium-a-comprehensive-framework-for","slug":"aquarium-a-comprehensive-framework-for","title":"Aquarium: A Comprehensive Framework for Exploring Predator-Prey Dynamics through Multi-Agent Reinforcement Learning Algorithms","date":"2024-01-13","arxiv_id":"2401.07056","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-trajectory-constrained-exploration","slug":"adaptive-trajectory-constrained-exploration","title":"Adaptive trajectory-constrained exploration strategy for deep reinforcement learning","date":"2023-12-27","arxiv_id":"2312.16456","repositories_listed":1,"syntology":null},{"url":"/paper/context-aware-communication-for-multi-agent","slug":"context-aware-communication-for-multi-agent","title":"Context-aware Communication for Multi-agent Reinforcement Learning","date":"2023-12-25","arxiv_id":"2312.15600","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-using-echo","slug":"multi-agent-reinforcement-learning-using-echo","title":"Multi-agent reinforcement learning using echo-state network and its application to pedestrian dynamics","date":"2023-12-19","arxiv_id":"2312.11834","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-via-1","slug":"multi-agent-reinforcement-learning-via-1","title":"Multi-Agent Reinforcement Learning via Distributed MPC as a Function Approximator","date":"2023-12-08","arxiv_id":"2312.05166","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarl-benchmarking-multi-agent","slug":"benchmarl-benchmarking-multi-agent","title":"BenchMARL: Benchmarking Multi-Agent Reinforcement Learning","date":"2023-12-03","arxiv_id":"2312.01472","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarl-benchmarking-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2312.01472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01472"}},"official":{"repos":["facebookresearch/benchmarl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-the-team-orienteering-problem-with-1","slug":"solving-the-team-orienteering-problem-with-1","title":"TOP-Former: A Multi-Agent Transformer Approach for the Team Orienteering Problem","date":"2023-11-30","arxiv_id":"2311.18662","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-quantum-reinforcement-learning","slug":"multi-agent-quantum-reinforcement-learning","title":"Multi-Agent Quantum Reinforcement Learning using Evolutionary Optimization","date":"2023-11-09","arxiv_id":"2311.05546","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-quantum-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2311.05546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05546"}},"official":{"repos":["michaelkoelle/qmarl-evo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/riskq-risk-sensitive-multi-agent","slug":"riskq-risk-sensitive-multi-agent","title":"RiskQ: Risk-sensitive Multi-Agent Reinforcement Learning Value Factorization","date":"2023-11-03","arxiv_id":"2311.01753","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/riskq-risk-sensitive-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2311.01753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01753"}},"official":{"repos":["xmu-rl-3dv/riskq"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/selectively-sharing-experiences-improves","slug":"selectively-sharing-experiences-improves","title":"Selectively Sharing Experiences Improves Multi-Agent Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00865","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/selectively-sharing-experiences-improves#ran","syntology_url":"https://syntology.ai/paper/2311.00865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.00865"}},"official":{"repos":["mgerstgrasser/super"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uav-pathfinding-in-dynamic-obstacle-avoidance","slug":"uav-pathfinding-in-dynamic-obstacle-avoidance","title":"Multi-Agent Reinforcement Learning-Based UAV Pathfinding for Obstacle Avoidance in Stochastic Environment","date":"2023-10-25","arxiv_id":"2310.16659","repositories_listed":1,"syntology":null},{"url":"/paper/depaint-a-decentralized-safe-multi-agent","slug":"depaint-a-decentralized-safe-multi-agent","title":"DePAint: A Decentralized Safe Multi-Agent Reinforcement Learning Algorithm considering Peak and Average Constraints","date":"2023-10-22","arxiv_id":"2310.14348","repositories_listed":1,"syntology":null},{"url":"/paper/theory-of-mind-for-multi-agent-collaboration","slug":"theory-of-mind-for-multi-agent-collaboration","title":"Theory of Mind for Multi-Agent Collaboration via Large Language Models","date":"2023-10-16","arxiv_id":"2310.10701","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-neuron-segmentation-with","slug":"self-supervised-neuron-segmentation-with","title":"Self-Supervised Neuron Segmentation with Multi-Agent Reinforcement Learning","date":"2023-10-06","arxiv_id":"2310.04148","repositories_listed":1,"syntology":{"n":24,"n_ran":20,"n_constructed":4,"n_ran_checked":18,"n_instrument":2,"n_unverified":4,"n_honours":2,"n_violates":1,"n_no_contract":15,"n_pointer_only":24,"phrase":"20 ran (of which 4 constructed an object rather than computing a result; 18 with no instrument failure: 2 honoured, 1 violated, 15 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/self-supervised-neuron-segmentation-with#ran","syntology_url":"https://syntology.ai/paper/2310.04148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04148"}},"official":{"repos":["ydchen0806/dbmim"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":4,"n_ran_no_instrument_failure":18,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-for-power","slug":"multi-agent-reinforcement-learning-for-power","title":"Multi-Agent Reinforcement Learning for Power Grid Topology Optimization","date":"2023-10-04","arxiv_id":"2310.02605","repositories_listed":1,"syntology":null},{"url":"/paper/cooperation-dynamics-in-multi-agent-systems","slug":"cooperation-dynamics-in-multi-agent-systems","title":"Cooperation Dynamics in Multi-Agent Systems: Exploring Game-Theoretic Scenarios with Mean-Field Equilibria","date":"2023-09-28","arxiv_id":"2309.16263","repositories_listed":1,"syntology":null},{"url":"/paper/effective-multi-agent-deep-reinforcement","slug":"effective-multi-agent-deep-reinforcement","title":"Effective Multi-Agent Deep Reinforcement Learning Control with Relative Entropy Regularization","date":"2023-09-26","arxiv_id":"2309.14727","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-conservative-q-learning-for-1","slug":"counterfactual-conservative-q-learning-for-1","title":"Counterfactual Conservative Q Learning for Offline Multi-agent Reinforcement Learning","date":"2023-09-22","arxiv_id":"2309.12696","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":3,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":9,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/counterfactual-conservative-q-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2309.12696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12696"}},"official":{"repos":["thu-rllab/CFCQL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/learning-zero-sum-linear-quadratic-games-with","slug":"learning-zero-sum-linear-quadratic-games-with","title":"Learning Zero-Sum Linear Quadratic Games with Improved Sample Complexity and Last-Iterate Convergence","date":"2023-09-08","arxiv_id":"2309.04272","repositories_listed":1,"syntology":null},{"url":"/paper/learning-collaborative-information","slug":"learning-collaborative-information","title":"Collaborative Information Dissemination with Graph-based Multi-Agent Reinforcement Learning","date":"2023-08-25","arxiv_id":"2308.16198","repositories_listed":1,"syntology":null},{"url":"/paper/molopt-autonomous-molecular-geometry","slug":"molopt-autonomous-molecular-geometry","title":"MolOpt: Autonomous Molecular Geometry Optimization using Multi-Agent Reinforcement Learning","date":"2023-08-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rm-e-3-equivariant-actor-critic-methods-for","slug":"rm-e-3-equivariant-actor-critic-methods-for","title":"${\\rm E}(3)$-Equivariant Actor-Critic Methods for Cooperative Multi-Agent Reinforcement Learning","date":"2023-08-23","arxiv_id":"2308.11842","repositories_listed":1,"syntology":null},{"url":"/paper/fox-formation-aware-exploration-in-multi","slug":"fox-formation-aware-exploration-in-multi","title":"FoX: Formation-aware exploration in multi-agent reinforcement learning","date":"2023-08-22","arxiv_id":"2308.11272","repositories_listed":1,"syntology":null},{"url":"/paper/comix-a-multi-agent-reinforcement-learning","slug":"comix-a-multi-agent-reinforcement-learning","title":"CoMIX: A Multi-agent Reinforcement Learning Training Architecture for Efficient Decentralized Coordination and Independent Decision-Making","date":"2023-08-21","arxiv_id":"2308.10721","repositories_listed":1,"syntology":null},{"url":"/paper/towards-few-shot-coordination-revisiting-ad","slug":"towards-few-shot-coordination-revisiting-ad","title":"Towards Few-shot Coordination: Revisiting Ad-hoc Teamplay Challenge In the Game of Hanabi","date":"2023-08-20","arxiv_id":"2308.10284","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-few-shot-coordination-revisiting-ad#ran","syntology_url":"https://syntology.ai/paper/2308.10284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10284"}},"official":{"repos":["chandar-lab/adaptive-hanabi"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dpmac-differentially-private-communication","slug":"dpmac-differentially-private-communication","title":"DPMAC: Differentially Private Communication for Cooperative Multi-Agent Reinforcement Learning","date":"2023-08-19","arxiv_id":"2308.09902","repositories_listed":1,"syntology":null},{"url":"/paper/heterogeneous-multi-agent-reinforcement-1","slug":"heterogeneous-multi-agent-reinforcement-1","title":"Heterogeneous Multi-Agent Reinforcement Learning via Mirror Descent Policy Optimization","date":"2023-08-13","arxiv_id":"2308.06741","repositories_listed":1,"syntology":null},{"url":"/paper/minimizing-return-gaps-with-discrete","slug":"minimizing-return-gaps-with-discrete","title":"RGMComm: Return Gap Minimization via Discrete Communications in Multi-Agent Reinforcement Learning","date":"2023-08-07","arxiv_id":"2308.03358","repositories_listed":1,"syntology":null},{"url":"/paper/communication-efficient-decentralized-multi","slug":"communication-efficient-decentralized-multi","title":"Communication-Efficient Decentralized Multi-Agent Reinforcement Learning for Cooperative Adaptive Cruise Control","date":"2023-08-04","arxiv_id":"2308.02345","repositories_listed":1,"syntology":null},{"url":"/paper/robust-multi-agent-reinforcement-learning-3","slug":"robust-multi-agent-reinforcement-learning-3","title":"Robust Multi-Agent Reinforcement Learning with State Uncertainty","date":"2023-07-30","arxiv_id":"2307.16212","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/robust-multi-agent-reinforcement-learning-3#ran","syntology_url":"https://syntology.ai/paper/2307.16212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16212"}},"official":{"repos":["sihongho/robust_marl_with_state_uncertainty"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-multi-agent-reinforcement-learning-1","slug":"offline-multi-agent-reinforcement-learning-1","title":"Offline Multi-Agent Reinforcement Learning with Implicit Global-to-Local Value Regularization","date":"2023-07-21","arxiv_id":"2307.11620","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/offline-multi-agent-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2307.11620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11620"}},"official":{"repos":["zhengyinan-air/omiga"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/viser-a-tractable-solution-concept-for-games","slug":"viser-a-tractable-solution-concept-for-games","title":"VISER: A Tractable Solution Concept for Games with Information Asymmetry","date":"2023-07-18","arxiv_id":"2307.09652","repositories_listed":1,"syntology":null},{"url":"/paper/sacha-soft-actor-critic-with-heuristic-based","slug":"sacha-soft-actor-critic-with-heuristic-based","title":"SACHA: Soft Actor-Critic with Heuristic-Based Attention for Partially Observable Multi-Agent Path Finding","date":"2023-07-05","arxiv_id":"2307.02691","repositories_listed":1,"syntology":null},{"url":"/paper/environmental-effects-on-emergent-strategy-in","slug":"environmental-effects-on-emergent-strategy-in","title":"Environmental effects on emergent strategy in micro-scale multi-agent reinforcement learning","date":"2023-07-03","arxiv_id":"2307.00994","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-causality-for-efficient","slug":"discovering-causality-for-efficient","title":"Discovering Causality for Efficient Cooperation in Multi-Agent Environments","date":"2023-06-20","arxiv_id":"2306.11846","repositories_listed":1,"syntology":null},{"url":"/paper/imp-marl-a-suite-of-environments-for-large","slug":"imp-marl-a-suite-of-environments-for-large","title":"IMP-MARL: a Suite of Environments for Large-scale Infrastructure Management Planning via MARL","date":"2023-06-20","arxiv_id":"2306.11551","repositories_listed":1,"syntology":null},{"url":"/paper/cammarl-conformal-action-modeling-in-multi","slug":"cammarl-conformal-action-modeling-in-multi","title":"CAMMARL: Conformal Action Modeling in Multi Agent Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.11128","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/cammarl-conformal-action-modeling-in-multi#ran","syntology_url":"https://syntology.ai/paper/2306.11128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11128"}},"official":{"repos":["nikunj-gupta/conformal-agent-modelling"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/maximum-entropy-heterogeneous-agent-mirror","slug":"maximum-entropy-heterogeneous-agent-mirror","title":"Maximum Entropy Heterogeneous-Agent Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.10715","repositories_listed":1,"syntology":null},{"url":"/paper/decentralized-social-navigation-with-non","slug":"decentralized-social-navigation-with-non","title":"Decentralized Social Navigation with Non-Cooperative Robots via Bi-Level Optimization","date":"2023-06-15","arxiv_id":"2306.08815","repositories_listed":1,"syntology":null},{"url":"/paper/mediated-multi-agent-reinforcement-learning","slug":"mediated-multi-agent-reinforcement-learning","title":"Mediated Multi-Agent Reinforcement Learning","date":"2023-06-14","arxiv_id":"2306.08419","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mediated-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2306.08419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08419"}},"official":{"repos":["dimonenka/mediatedmarl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-versatile-multi-agent-reinforcement","slug":"a-versatile-multi-agent-reinforcement","title":"A Versatile Multi-Agent Reinforcement Learning Benchmark for Inventory Management","date":"2023-06-13","arxiv_id":"2306.07542","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-versatile-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.07542","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07542"}},"official":{"repos":["victoryxl/replenishmentenv"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/iplan-intent-aware-planning-in-heterogeneous","slug":"iplan-intent-aware-planning-in-heterogeneous","title":"iPLAN: Intent-Aware Planning in Heterogeneous Traffic via Distributed Multi-Agent Reinforcement Learning","date":"2023-06-09","arxiv_id":"2306.06236","repositories_listed":1,"syntology":null},{"url":"/paper/progression-cognition-reinforcement-learning","slug":"progression-cognition-reinforcement-learning","title":"Progression Cognition Reinforcement Learning with Prioritized Experience for Multi-Vehicle Pursuit","date":"2023-06-08","arxiv_id":"2306.05016","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-framework-for-factorizing","slug":"a-unified-framework-for-factorizing","title":"A Unified Framework for Factorizing Distributional Value Functions for Multi-Agent Reinforcement Learning","date":"2023-06-04","arxiv_id":"2306.02430","repositories_listed":1,"syntology":null},{"url":"/paper/ma2cl-masked-attentive-contrastive-learning","slug":"ma2cl-masked-attentive-contrastive-learning","title":"MA2CL:Masked Attentive Contrastive Learning for Multi-Agent Reinforcement Learning","date":"2023-06-03","arxiv_id":"2306.02006","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":4,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ma2cl-masked-attentive-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2306.02006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02006"}},"official":{"repos":["ustchlsong/ma2cl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/model-aided-federated-reinforcement-learning","slug":"model-aided-federated-reinforcement-learning","title":"Model-aided Federated Reinforcement Learning for Multi-UAV Trajectory Planning in IoT Networks","date":"2023-06-03","arxiv_id":"2306.02029","repositories_listed":1,"syntology":null},{"url":"/paper/context-aware-bayesian-network-actor-critic","slug":"context-aware-bayesian-network-actor-critic","title":"Context-Aware Bayesian Network Actor-Critic Methods for Cooperative Multi-Agent Reinforcement Learning","date":"2023-06-02","arxiv_id":"2306.01920","repositories_listed":1,"syntology":null},{"url":"/paper/relu-to-the-rescue-improve-your-on-policy","slug":"relu-to-the-rescue-improve-your-on-policy","title":"ReLU to the Rescue: Improve Your On-Policy Actor-Critic with Positive Advantages","date":"2023-06-02","arxiv_id":"2306.01460","repositories_listed":1,"syntology":null},{"url":"/paper/expode-exploiting-policy-discrepancy-for","slug":"expode-exploiting-policy-discrepancy-for","title":"EXPODE: EXploiting POlicy Discrepancy for Efficient Exploration in Multi-agent Reinforcement Learning","date":"2023-05-30","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/is-centralized-training-with-decentralized","slug":"is-centralized-training-with-decentralized","title":"Is Centralized Training with Decentralized Execution Framework Centralized Enough for MARL?","date":"2023-05-27","arxiv_id":"2305.17352","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/is-centralized-training-with-decentralized#ran","syntology_url":"https://syntology.ai/paper/2305.17352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17352"}},"official":{"repos":["zyh1999/cadp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/explainable-multi-agent-reinforcement","slug":"explainable-multi-agent-reinforcement","title":"Explainable Multi-Agent Reinforcement Learning for Temporal Queries","date":"2023-05-17","arxiv_id":"2305.10378","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-on-google-research","slug":"an-empirical-study-on-google-research","title":"An Empirical Study on Google Research Football Multi-agent Scenarios","date":"2023-05-16","arxiv_id":"2305.09458","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-resources","slug":"multi-agent-reinforcement-learning-resources","title":"Multi-Agent Reinforcement Learning Resources Allocation Method Using Dueling Double Deep Q-Network in Vehicular Networks","date":"2023-05-12","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/robust-multi-agent-coordination-via","slug":"robust-multi-agent-coordination-via","title":"Robust multi-agent coordination via evolutionary generation of auxiliary adversarial attackers","date":"2023-05-10","arxiv_id":"2305.05909","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/robust-multi-agent-coordination-via#ran","syntology_url":"https://syntology.ai/paper/2305.05909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.05909"}},"official":{"repos":["zzq-bot/romance"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/smaclite-a-lightweight-environment-for-multi","slug":"smaclite-a-lightweight-environment-for-multi","title":"SMAClite: A Lightweight Environment for Multi-Agent Reinforcement Learning","date":"2023-05-09","arxiv_id":"2305.05566","repositories_listed":1,"syntology":null},{"url":"/paper/information-design-in-multi-agent","slug":"information-design-in-multi-agent","title":"Information Design in Multi-Agent Reinforcement Learning","date":"2023-05-08","arxiv_id":"2305.06807","repositories_listed":1,"syntology":null},{"url":"/paper/local-optimization-achieves-global-optimality","slug":"local-optimization-achieves-global-optimality","title":"Local Optimization Achieves Global Optimality in Multi-Agent Reinforcement Learning","date":"2023-05-08","arxiv_id":"2305.04819","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/local-optimization-achieves-global-optimality#ran","syntology_url":"https://syntology.ai/paper/2305.04819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04819"}},"official":{"repos":["zhaoyl18/ratio_game"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/system-neural-diversity-measuring-behavioral","slug":"system-neural-diversity-measuring-behavioral","title":"System Neural Diversity: Measuring Behavioral Heterogeneity in Multi-Agent Learning","date":"2023-05-03","arxiv_id":"2305.02128","repositories_listed":1,"syntology":null},{"url":"/paper/centralized-control-for-multi-agent-rl-in-a","slug":"centralized-control-for-multi-agent-rl-in-a","title":"Centralized control for multi-agent RL in a complex Real-Time-Strategy game","date":"2023-04-25","arxiv_id":"2304.13004","repositories_listed":1,"syntology":null},{"url":"/paper/partially-observable-mean-field-multi-agent","slug":"partially-observable-mean-field-multi-agent","title":"Partially Observable Mean Field Multi-Agent Reinforcement Learning Based on Graph-Attention","date":"2023-04-25","arxiv_id":"2304.12653","repositories_listed":1,"syntology":null},{"url":"/paper/stubborn-an-environment-for-evaluating","slug":"stubborn-an-environment-for-evaluating","title":"Stubborn: An Environment for Evaluating Stubbornness between Agents with Aligned Incentives","date":"2023-04-24","arxiv_id":"2304.12280","repositories_listed":1,"syntology":null},{"url":"/paper/heterogeneous-agent-reinforcement-learning","slug":"heterogeneous-agent-reinforcement-learning","title":"Heterogeneous-Agent Reinforcement Learning","date":"2023-04-19","arxiv_id":"2304.09870","repositories_listed":1,"syntology":null},{"url":"/paper/stas-spatial-temporal-return-decomposition","slug":"stas-spatial-temporal-return-decomposition","title":"STAS: Spatial-Temporal Return Decomposition for Multi-agent Reinforcement Learning","date":"2023-04-15","arxiv_id":"2304.07520","repositories_listed":1,"syntology":null},{"url":"/paper/language-instructed-reinforcement-learning","slug":"language-instructed-reinforcement-learning","title":"Language Instructed Reinforcement Learning for Human-AI Coordination","date":"2023-04-13","arxiv_id":"2304.07297","repositories_listed":1,"syntology":null},{"url":"/paper/marl-idr-multi-agent-reinforcement-learning","slug":"marl-idr-multi-agent-reinforcement-learning","title":"MARL-iDR: Multi-Agent Reinforcement Learning for Incentive-based Residential Demand Response","date":"2023-04-08","arxiv_id":"2304.04086","repositories_listed":1,"syntology":null},{"url":"/paper/effective-control-of-two-dimensional-rayleigh","slug":"effective-control-of-two-dimensional-rayleigh","title":"Effective control of two-dimensional Rayleigh--Bénard convection: invariant multi-agent reinforcement learning is all you need","date":"2023-04-05","arxiv_id":"2304.02370","repositories_listed":1,"syntology":null},{"url":"/paper/effective-and-stable-role-based-multi-agent","slug":"effective-and-stable-role-based-multi-agent","title":"Effective and Stable Role-Based Multi-Agent Collaboration by Structural Information Principles","date":"2023-04-03","arxiv_id":"2304.00755","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":4,"n_ran_checked":5,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":11,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/effective-and-stable-role-based-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2304.00755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.00755"}},"official":{"repos":["ringbdstack/sr-marl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-with-6","slug":"multi-agent-reinforcement-learning-with-6","title":"Multi-Agent Reinforcement Learning with Action Masking for UAV-enabled Mobile Communications","date":"2023-03-29","arxiv_id":"2303.16737","repositories_listed":1,"syntology":null},{"url":"/paper/marl-jax-multi-agent-reinforcement-leaning","slug":"marl-jax-multi-agent-reinforcement-leaning","title":"marl-jax: Multi-Agent Reinforcement Leaning Framework","date":"2023-03-24","arxiv_id":"2303.13808","repositories_listed":1,"syntology":null},{"url":"/paper/conditionally-optimistic-exploration-for","slug":"conditionally-optimistic-exploration-for","title":"Conditionally Optimistic Exploration for Cooperative Deep Multi-Agent Reinforcement Learning","date":"2023-03-16","arxiv_id":"2303.09032","repositories_listed":1,"syntology":null},{"url":"/paper/decentralized-multi-agent-reinforcement-4","slug":"decentralized-multi-agent-reinforcement-4","title":"Decentralized Multi-Agent Reinforcement Learning for Continuous-Space Stochastic Games","date":"2023-03-16","arxiv_id":"2303.13539","repositories_listed":1,"syntology":null},{"url":"/paper/mahtm-a-multi-agent-framework-for","slug":"mahtm-a-multi-agent-framework-for","title":"MAHTM: A Multi-Agent Framework for Hierarchical Transactive Microgrids","date":"2023-03-15","arxiv_id":"2303.08447","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/mahtm-a-multi-agent-framework-for#ran","syntology_url":"https://syntology.ai/paper/2303.08447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08447"}},"official":{"repos":["nicosquare/rl-energy-management"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/poseexaminer-automated-testing-of-out-of","slug":"poseexaminer-automated-testing-of-out-of","title":"PoseExaminer: Automated Testing of Out-of-Distribution Robustness in Human Pose and Shape Estimation","date":"2023-03-13","arxiv_id":"2303.07337","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/poseexaminer-automated-testing-of-out-of#ran","syntology_url":"https://syntology.ai/paper/2303.07337","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07337"}},"official":{"repos":["qihao067/poseexaminer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/socialgym-2-0-simulator-for-multi-agent","slug":"socialgym-2-0-simulator-for-multi-agent","title":"SOCIALGYM 2.0: Simulator for Multi-Agent Social Robot Navigation in Shared Human Spaces","date":"2023-03-09","arxiv_id":"2303.05584","repositories_listed":1,"syntology":null},{"url":"/paper/solving-routing-problems-for-multiple","slug":"solving-routing-problems-for-multiple","title":"Solving routing problems for multiple cooperative Unmanned Aerial Vehicles using Transformer networks, vol. 122, pp. 106085, 2023","date":"2023-03-09","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ghq-grouped-hybrid-q-learning-for","slug":"ghq-grouped-hybrid-q-learning-for","title":"GHQ: Grouped Hybrid Q Learning for Heterogeneous Cooperative Multi-agent Reinforcement Learning","date":"2023-03-02","arxiv_id":"2303.01070","repositories_listed":1,"syntology":null},{"url":"/paper/iq-flow-mechanism-design-for-inducing","slug":"iq-flow-mechanism-design-for-inducing","title":"IQ-Flow: Mechanism Design for Inducing Cooperative Behavior to Self-Interested Agents in Sequential Social Dilemmas","date":"2023-02-28","arxiv_id":"2302.14604","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-the-gumbel-softmax-in-maddpg","slug":"revisiting-the-gumbel-softmax-in-maddpg","title":"Revisiting the Gumbel-Softmax in MADDPG","date":"2023-02-23","arxiv_id":"2302.11793","repositories_listed":1,"syntology":null},{"url":"/paper/mac-po-multi-agent-experience-replay-via","slug":"mac-po-multi-agent-experience-replay-via","title":"MAC-PO: Multi-Agent Experience Replay via Collective Priority Optimization","date":"2023-02-21","arxiv_id":"2302.10418","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-value-decomposition-with-greedy","slug":"adaptive-value-decomposition-with-greedy","title":"Adaptive Value Decomposition with Greedy Marginal Contribution Computation for Cooperative Multi-Agent Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.06872","repositories_listed":1,"syntology":null},{"url":"/paper/learning-complex-teamwork-tasks-using-a-sub","slug":"learning-complex-teamwork-tasks-using-a-sub","title":"Learning Complex Teamwork Tasks Using a Given Sub-task Decomposition","date":"2023-02-09","arxiv_id":"2302.04944","repositories_listed":1,"syntology":null},{"url":"/paper/learning-graph-enhanced-commander-executor","slug":"learning-graph-enhanced-commander-executor","title":"Learning Graph-Enhanced Commander-Executor for Multi-Agent Navigation","date":"2023-02-08","arxiv_id":"2302.04094","repositories_listed":1,"syntology":null},{"url":"/paper/policy-evaluation-in-decentralized-pomdps","slug":"policy-evaluation-in-decentralized-pomdps","title":"Policy Evaluation in Decentralized POMDPs with Belief Sharing","date":"2023-02-08","arxiv_id":"2302.04151","repositories_listed":1,"syntology":null},{"url":"/paper/attacking-cooperative-multi-agent","slug":"attacking-cooperative-multi-agent","title":"Attacking Cooperative Multi-Agent Reinforcement Learning by Adversarial Minority Influence","date":"2023-02-07","arxiv_id":"2302.03322","repositories_listed":1,"syntology":null},{"url":"/paper/uncoupled-learning-of-differential","slug":"uncoupled-learning-of-differential","title":"Uncoupled Learning of Differential Stackelberg Equilibria with Commitments","date":"2023-02-07","arxiv_id":"2302.03438","repositories_listed":1,"syntology":null},{"url":"/paper/learning-zero-shot-cooperation-with-humans","slug":"learning-zero-shot-cooperation-with-humans","title":"Learning Zero-Shot Cooperation with Humans, Assuming Humans Are Biased","date":"2023-02-03","arxiv_id":"2302.01605","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-zero-shot-cooperation-with-humans#ran","syntology_url":"https://syntology.ai/paper/2302.01605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.01605"}},"official":{"repos":["samjia2000/HSP"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-multiple-independent-advisors","slug":"learning-from-multiple-independent-advisors","title":"Learning from Multiple Independent Advisors in Multi-agent Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.11153","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-congestion-cost-minimization-with","slug":"multi-agent-congestion-cost-minimization-with","title":"Multi-Agent Congestion Cost Minimization With Linear Function Approximations","date":"2023-01-26","arxiv_id":"2301.10993","repositories_listed":1,"syntology":null}],"record_sha256":"7b3406860e77611a3f8719d05ff72f1ad78d7378714948e4972760da73bc2c81","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}