{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multi-agent-reinforcement-learning/papers/4","list_of":"/task/multi-agent-reinforcement-learning","task":"Multi-agent Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":18,"rows_per_page":100,"rows":[301,400],"of":1718,"counts":{"archive_papers_tagged":1718,"with_a_code_link":522,"where_syntology_ran_a_sample":135,"not_listed_spam_title":0,"listed":1718,"listed_where_code_ran":135,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":119,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":119,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multi-agent-reinforcement-learning","prev":"/task/multi-agent-reinforcement-learning/papers/3","next":"/task/multi-agent-reinforcement-learning/papers/5","papers":[{"url":"/paper/investigating-the-impact-of-direct-punishment","slug":"investigating-the-impact-of-direct-punishment","title":"Investigating the Impact of Direct Punishment on the Emergence of Cooperation in Multi-Agent Reinforcement Learning Systems","date":"2023-01-19","arxiv_id":"2301.08278","repositories_listed":1,"syntology":null},{"url":"/paper/mean-field-control-based-approximation-of","slug":"mean-field-control-based-approximation-of","title":"Mean-Field Control based Approximation of Multi-Agent Reinforcement Learning in Presence of a Non-decomposable Shared Global State","date":"2023-01-13","arxiv_id":"2301.06889","repositories_listed":1,"syntology":null},{"url":"/paper/transfqmix-transformers-for-leveraging-the","slug":"transfqmix-transformers-for-leveraging-the","title":"TransfQMix: Transformers for Leveraging the Graph Structure of Multi-Agent Reinforcement Learning Problems","date":"2023-01-13","arxiv_id":"2301.05334","repositories_listed":1,"syntology":null},{"url":"/paper/self-motivated-multi-agent-exploration","slug":"self-motivated-multi-agent-exploration","title":"Self-Motivated Multi-Agent Exploration","date":"2023-01-05","arxiv_id":"2301.02083","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-robustness-assessment-via","slug":"efficient-robustness-assessment-via","title":"Efficient Robustness Assessment via Adversarial Spatial-Temporal Focus on Videos","date":"2023-01-03","arxiv_id":"2301.00896","repositories_listed":1,"syntology":null},{"url":"/paper/strangeness-driven-exploration-in-multi-agent","slug":"strangeness-driven-exploration-in-multi-agent","title":"Strangeness-driven Exploration in Multi-Agent Reinforcement Learning","date":"2022-12-27","arxiv_id":"2212.13448","repositories_listed":1,"syntology":null},{"url":"/paper/certified-policy-smoothing-for-cooperative","slug":"certified-policy-smoothing-for-cooperative","title":"Certified Policy Smoothing for Cooperative Multi-Agent Reinforcement Learning","date":"2022-12-22","arxiv_id":"2212.11746","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-multi-agent-reinforcement-learning-3","slug":"scalable-multi-agent-reinforcement-learning-3","title":"Scalable Multi-Agent Reinforcement Learning for Warehouse Logistics with Robotic and Human Co-Workers","date":"2022-12-22","arxiv_id":"2212.11498","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-multi-agent-reinforcement-learning-3#ran","syntology_url":"https://syntology.ai/paper/2212.11498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.11498"}},"official":{"repos":["uoe-agents/task-assignment-robotic-warehouse"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/distributed-training-and-execution-multi","slug":"distributed-training-and-execution-multi","title":"Distributed-Training-and-Execution Multi-Agent Reinforcement Learning for Power Control in HetNet","date":"2022-12-15","arxiv_id":"2212.07967","repositories_listed":1,"syntology":null},{"url":"/paper/smacv2-an-improved-benchmark-for-cooperative-1","slug":"smacv2-an-improved-benchmark-for-cooperative-1","title":"SMACv2: An Improved Benchmark for Cooperative Multi-Agent Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07489","repositories_listed":1,"syntology":null},{"url":"/paper/effects-of-spectral-normalization-in-multi","slug":"effects-of-spectral-normalization-in-multi","title":"Effects of Spectral Normalization in Multi-agent Reinforcement Learning","date":"2022-12-10","arxiv_id":"2212.05331","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/effects-of-spectral-normalization-in-multi#ran","syntology_url":"https://syntology.ai/paper/2212.05331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05331"}},"official":{"repos":["kinalmehta/epymarl_spectral"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-is-the-solution-for-state-adversarial","slug":"what-is-the-solution-for-state-adversarial","title":"What is the Solution for State-Adversarial Multi-Agent Reinforcement Learning?","date":"2022-12-06","arxiv_id":"2212.02705","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/what-is-the-solution-for-state-adversarial#ran","syntology_url":"https://syntology.ai/paper/2212.02705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02705"}},"official":{"repos":["susanbao/rmarl_code"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/ace-cooperative-multi-agent-q-learning-with","slug":"ace-cooperative-multi-agent-q-learning-with","title":"ACE: Cooperative Multi-agent Q-learning with Bidirectional Action-Dependency","date":"2022-11-29","arxiv_id":"2211.16068","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ace-cooperative-multi-agent-q-learning-with#ran","syntology_url":"https://syntology.ai/paper/2211.16068","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.16068"}},"official":{"repos":["opendilab/ace"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contrastive-identity-aware-learning-for-multi","slug":"contrastive-identity-aware-learning-for-multi","title":"Contrastive Identity-Aware Learning for Multi-Agent Value Decomposition","date":"2022-11-23","arxiv_id":"2211.12712","repositories_listed":1,"syntology":null},{"url":"/paper/tinyqmix-distributed-access-control-for-mmtc","slug":"tinyqmix-distributed-access-control-for-mmtc","title":"TinyQMIX: Distributed Access Control for mMTC via Multi-agent Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11692","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-action-advising-for-multi-agent","slug":"explainable-action-advising-for-multi-agent","title":"Explainable Action Advising for Multi-Agent Reinforcement Learning","date":"2022-11-15","arxiv_id":"2211.07882","repositories_listed":1,"syntology":null},{"url":"/paper/fleet-rebalancing-for-expanding-shared-e","slug":"fleet-rebalancing-for-expanding-shared-e","title":"Fleet Rebalancing for Expanding Shared e-Mobility Systems: A Multi-agent Deep Reinforcement Learning Approach","date":"2022-11-11","arxiv_id":"2211.06136","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-13","slug":"multi-agent-reinforcement-learning-for-13","title":"Multi-Agent Reinforcement Learning for Adaptive Mesh Refinement","date":"2022-11-02","arxiv_id":"2211.00801","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-13#ran","syntology_url":"https://syntology.ai/paper/2211.00801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00801"}},"official":{"repos":["011235813/marl-amr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/agent-time-attention-for-sparse-rewards-multi","slug":"agent-time-attention-for-sparse-rewards-multi","title":"Agent-Time Attention for Sparse Rewards Multi-Agent Reinforcement Learning","date":"2022-10-31","arxiv_id":"2210.17540","repositories_listed":1,"syntology":null},{"url":"/paper/teal-learning-accelerated-optimization-of","slug":"teal-learning-accelerated-optimization-of","title":"Teal: Learning-Accelerated Optimization of WAN Traffic Engineering","date":"2022-10-25","arxiv_id":"2210.13763","repositories_listed":1,"syntology":null},{"url":"/paper/idrl-identifying-identities-in-multi-agent","slug":"idrl-identifying-identities-in-multi-agent","title":"Classifying Ambiguous Identities in Hidden-Role Stochastic Games with Multi-Agent Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.12896","repositories_listed":1,"syntology":null},{"url":"/paper/solving-continuous-control-via-q-learning","slug":"solving-continuous-control-via-q-learning","title":"Solving Continuous Control via Q-learning","date":"2022-10-22","arxiv_id":"2210.12566","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/solving-continuous-control-via-q-learning#ran","syntology_url":"https://syntology.ai/paper/2210.12566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.12566"}},"official":{"repos":["tseyde/decqn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/proximal-learning-with-opponent-learning","slug":"proximal-learning-with-opponent-learning","title":"Proximal Learning With Opponent-Learning Awareness","date":"2022-10-18","arxiv_id":"2210.10125","repositories_listed":1,"syntology":{"n":20,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":13,"n_honours":5,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 5 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/proximal-learning-with-opponent-learning#ran","syntology_url":"https://syntology.ai/paper/2210.10125","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10125"}},"official":{"repos":["silent-zebra/pola"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":12,"ran_from_kinds":["official"]}}},{"url":"/paper/distributional-reward-estimation-for","slug":"distributional-reward-estimation-for","title":"Distributional Reward Estimation for Effective Multi-Agent Deep Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07636","repositories_listed":1,"syntology":null},{"url":"/paper/centralized-training-with-hybrid-execution-in","slug":"centralized-training-with-hybrid-execution-in","title":"Centralized Training with Hybrid Execution in Multi-Agent Reinforcement Learning","date":"2022-10-12","arxiv_id":"2210.06274","repositories_listed":1,"syntology":null},{"url":"/paper/phantom-an-rl-driven-framework-for-agent","slug":"phantom-an-rl-driven-framework-for-agent","title":"Phantom -- A RL-driven multi-agent framework to model complex systems","date":"2022-10-12","arxiv_id":"2210.06012","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/phantom-an-rl-driven-framework-for-agent#ran","syntology_url":"https://syntology.ai/paper/2210.06012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06012"}},"official":{"repos":["jpmorganchase/Phantom"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/digital-twin-based-multiple-access","slug":"digital-twin-based-multiple-access","title":"Digital Twin-Based Multiple Access Optimization and Monitoring via Model-Driven Bayesian Learning","date":"2022-10-11","arxiv_id":"2210.05582","repositories_listed":1,"syntology":null},{"url":"/paper/marllib-extending-rllib-for-multi-agent","slug":"marllib-extending-rllib-for-multi-agent","title":"MARLlib: A Scalable and Efficient Multi-agent Reinforcement Learning Library","date":"2022-10-11","arxiv_id":"2210.13708","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/marllib-extending-rllib-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2210.13708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13708"}},"official":{"repos":["replicable-marl/marllib"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-credit-assignment-for-cooperative","slug":"learning-credit-assignment-for-cooperative","title":"Learning Explicit Credit Assignment for Cooperative Multi-Agent Reinforcement Learning via Polarization Policy Gradient","date":"2022-10-10","arxiv_id":"2210.05367","repositories_listed":1,"syntology":null},{"url":"/paper/multiagent-reinforcement-learning-based-on","slug":"multiagent-reinforcement-learning-based-on","title":"Multiagent Reinforcement Learning Based on Fusion-Multiactor-Attention-Critic for Multiple-Unmanned-Aerial-Vehicle Navigation Control","date":"2022-10-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/elign-expectation-alignment-as-a-multi-agent","slug":"elign-expectation-alignment-as-a-multi-agent","title":"ELIGN: Expectation Alignment as a Multi-Agent Intrinsic Reward","date":"2022-10-09","arxiv_id":"2210.04365","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-laws-for-a-multi-agent-reinforcement","slug":"scaling-laws-for-a-multi-agent-reinforcement","title":"Scaling Laws for a Multi-Agent Reinforcement Learning Model","date":"2022-09-29","arxiv_id":"2210.00849","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-laws-for-a-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.00849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00849"}},"official":{"repos":["orenneumann/alphazero-scaling-laws"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pareto-actor-critic-for-equilibrium-selection","slug":"pareto-actor-critic-for-equilibrium-selection","title":"Pareto Actor-Critic for Equilibrium Selection in Multi-Agent Reinforcement Learning","date":"2022-09-28","arxiv_id":"2209.14344","repositories_listed":1,"syntology":null},{"url":"/paper/more-centralized-training-still-decentralized","slug":"more-centralized-training-still-decentralized","title":"More Centralized Training, Still Decentralized Execution: Multi-Agent Conditional Policy Factorization","date":"2022-09-26","arxiv_id":"2209.12681","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/more-centralized-training-still-decentralized#ran","syntology_url":"https://syntology.ai/paper/2209.12681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.12681"}},"official":{"repos":["pku-rl/fop-dmac-macpf"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/towards-a-standardised-performance-evaluation","slug":"towards-a-standardised-performance-evaluation","title":"Towards a Standardised Performance Evaluation Protocol for Cooperative MARL","date":"2022-09-21","arxiv_id":"2209.10485","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-a-standardised-performance-evaluation#ran","syntology_url":"https://syntology.ai/paper/2209.10485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.10485"}},"official":null}},{"url":"/paper/learning-sparse-graphon-mean-field-games","slug":"learning-sparse-graphon-mean-field-games","title":"Learning Sparse Graphon Mean Field Games","date":"2022-09-08","arxiv_id":"2209.03880","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-sparse-graphon-mean-field-games#ran","syntology_url":"https://syntology.ai/paper/2209.03880","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.03880"}},"official":{"repos":["chrfabian/learning_sparse_gmfgs"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-collaboration-of-multi-agent-model-using-an","slug":"a-collaboration-of-multi-agent-model-using-an","title":"A collaboration of multi-agent model using an interactive interface","date":"2022-09-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/get-it-in-writing-formal-contracts-mitigate","slug":"get-it-in-writing-formal-contracts-mitigate","title":"Formal Contracts Mitigate Social Dilemmas in Multi-Agent RL","date":"2022-08-22","arxiv_id":"2208.10469","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/get-it-in-writing-formal-contracts-mitigate#ran","syntology_url":"https://syntology.ai/paper/2208.10469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.10469"}},"official":{"repos":["algorithmic-alignment-lab/contracts"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/last-iterate-convergence-with-full-and-noisy","slug":"last-iterate-convergence-with-full-and-noisy","title":"Last-Iterate Convergence with Full and Noisy Feedback in Two-Player Zero-Sum Games","date":"2022-08-21","arxiv_id":"2208.09855","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/last-iterate-convergence-with-full-and-noisy#ran","syntology_url":"https://syntology.ai/paper/2208.09855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.09855"}},"official":{"repos":["cyberagentailab/m2wu"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/transformer-based-value-function","slug":"transformer-based-value-function","title":"Transformer-based Value Function Decomposition for Cooperative Multi-agent Reinforcement Learning in StarCraft","date":"2022-08-15","arxiv_id":"2208.07298","repositories_listed":1,"syntology":null},{"url":"/paper/heterogeneous-multi-agent-zero-shot","slug":"heterogeneous-multi-agent-zero-shot","title":"Heterogeneous Multi-agent Zero-Shot Coordination by Coevolution","date":"2022-08-09","arxiv_id":"2208.04957","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/heterogeneous-multi-agent-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2208.04957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.04957"}},"official":{"repos":["lamda-bbo/maze"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/interaction-pattern-disentangling-for-multi","slug":"interaction-pattern-disentangling-for-multi","title":"Interaction Pattern Disentangling for Multi-Agent Reinforcement Learning","date":"2022-07-08","arxiv_id":"2207.03902","repositories_listed":1,"syntology":null},{"url":"/paper/vmas-a-vectorized-multi-agent-simulator-for","slug":"vmas-a-vectorized-multi-agent-simulator-for","title":"VMAS: A Vectorized Multi-Agent Simulator for Collective Robot Learning","date":"2022-07-07","arxiv_id":"2207.03530","repositories_listed":1,"syntology":null},{"url":"/paper/learning-task-embeddings-for-teamwork","slug":"learning-task-embeddings-for-teamwork","title":"Learning Task Embeddings for Teamwork Adaptation in Multi-Agent Reinforcement Learning","date":"2022-07-05","arxiv_id":"2207.02249","repositories_listed":1,"syntology":null},{"url":"/paper/the-starcraft-multi-agent-challenges-learning","slug":"the-starcraft-multi-agent-challenges-learning","title":"The StarCraft Multi-Agent Challenges+ : Learning of Multi-Stage Tasks and Environmental Factors without Precise Reward Functions","date":"2022-07-05","arxiv_id":"2207.02007","repositories_listed":1,"syntology":null},{"url":"/paper/distspectrl-distributing-specifications-in","slug":"distspectrl-distributing-specifications-in","title":"DistSPECTRL: Distributing Specifications in Multi-Agent Reinforcement Learning Systems","date":"2022-06-28","arxiv_id":"2206.13754","repositories_listed":1,"syntology":null},{"url":"/paper/toward-multi-target-self-organizing-pursuit","slug":"toward-multi-target-self-organizing-pursuit","title":"Toward multi-target self-organizing pursuit in a partially observable Markov game","date":"2022-06-24","arxiv_id":"2206.12330","repositories_listed":1,"syntology":null},{"url":"/paper/pac-assisted-value-factorisation-with","slug":"pac-assisted-value-factorisation-with","title":"PAC: Assisted Value Factorisation with Counterfactual Predictions in Multi-Agent Reinforcement Learning","date":"2022-06-22","arxiv_id":"2206.11420","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/pac-assisted-value-factorisation-with#ran","syntology_url":"https://syntology.ai/paper/2206.11420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11420"}},"official":{"repos":["hanhananderson/pac-marl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/maser-multi-agent-reinforcement-learning-with","slug":"maser-multi-agent-reinforcement-learning-with","title":"MASER: Multi-Agent Reinforcement Learning with Subgoals Generated from Experience Replay Buffer","date":"2022-06-20","arxiv_id":"2206.10607","repositories_listed":1,"syntology":null},{"url":"/paper/logic-based-reward-shaping-for-multi-agent","slug":"logic-based-reward-shaping-for-multi-agent","title":"Logic-based Reward Shaping for Multi-Agent Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.08881","repositories_listed":1,"syntology":null},{"url":"/paper/universally-expressive-communication-in-multi","slug":"universally-expressive-communication-in-multi","title":"Universally Expressive Communication in Multi-Agent Reinforcement Learning","date":"2022-06-14","arxiv_id":"2206.06758","repositories_listed":1,"syntology":null},{"url":"/paper/stabilizing-voltage-in-power-distribution","slug":"stabilizing-voltage-in-power-distribution","title":"Stabilizing Voltage in Power Distribution Networks via Multi-Agent Reinforcement Learning with Transformer","date":"2022-06-08","arxiv_id":"2206.03721","repositories_listed":1,"syntology":null},{"url":"/paper/learning-distributed-and-fair-policies-for","slug":"learning-distributed-and-fair-policies-for","title":"Learning Distributed and Fair Policies for Network Load Balancing as Markov Potential Game","date":"2022-06-03","arxiv_id":"2206.01451","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-distributed-and-fair-policies-for#ran","syntology_url":"https://syntology.ai/paper/2206.01451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01451"}},"official":{"repos":["zhiyuanyaoj/marllb"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dm-2-distributed-multi-agent-reinforcement","slug":"dm-2-distributed-multi-agent-reinforcement","title":"DM$^2$: Decentralized Multi-Agent Reinforcement Learning for Distribution Matching","date":"2022-06-01","arxiv_id":"2206.00233","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-is-a","slug":"multi-agent-reinforcement-learning-is-a","title":"Multi-Agent Reinforcement Learning is a Sequence Modeling Problem","date":"2022-05-30","arxiv_id":"2205.14953","repositories_listed":1,"syntology":null},{"url":"/paper/alma-hierarchical-learning-for-composite","slug":"alma-hierarchical-learning-for-composite","title":"ALMA: Hierarchical Learning for Composite Multi-Agent Tasks","date":"2022-05-27","arxiv_id":"2205.14205","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/alma-hierarchical-learning-for-composite#ran","syntology_url":"https://syntology.ai/paper/2205.14205","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14205"}},"official":{"repos":["shariqiqbal2810/alma"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/qgnn-value-function-factorisation-with-graph","slug":"qgnn-value-function-factorisation-with-graph","title":"QGNN: Value Function Factorisation with Graph Neural Networks","date":"2022-05-25","arxiv_id":"2205.13005","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-multi-agent-model-based","slug":"scalable-multi-agent-model-based","title":"Scalable Multi-Agent Model-Based Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.15023","repositories_listed":1,"syntology":null},{"url":"/paper/self-paced-multi-agent-reinforcement-learning","slug":"self-paced-multi-agent-reinforcement-learning","title":"Learning Progress Driven Multi-Agent Curriculum","date":"2022-05-20","arxiv_id":"2205.10016","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-greedy-search-tracking-by-multi-agent","slug":"beyond-greedy-search-tracking-by-multi-agent","title":"Beyond Greedy Search: Tracking by Multi-Agent Reinforcement Learning-based Beam Search","date":"2022-05-19","arxiv_id":"2205.09676","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-transmission-control-for-wireless","slug":"distributed-transmission-control-for-wireless","title":"Distributed Transmission Control for Wireless Networks using Multi-Agent Reinforcement Learning","date":"2022-05-13","arxiv_id":"2205.06800","repositories_listed":1,"syntology":null},{"url":"/paper/conversational-ai-for-positive-sum-retailing","slug":"conversational-ai-for-positive-sum-retailing","title":"Conversational AI for Positive-sum Retailing under Falsehood Control","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/using-fuzzy-logic-to-learn-abstract-policies","slug":"using-fuzzy-logic-to-learn-abstract-policies","title":"Using Fuzzy Logic to Learn Abstract Policies in Large-Scale Multi-Agent Reinforcement Learning","date":"2022-04-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-11","slug":"multi-agent-reinforcement-learning-for-11","title":"Multi-Agent Reinforcement Learning for Traffic Signal Control through Universal Communication Method","date":"2022-04-26","arxiv_id":"2204.12190","repositories_listed":1,"syntology":null},{"url":"/paper/toward-policy-explanations-for-multi-agent","slug":"toward-policy-explanations-for-multi-agent","title":"Toward Policy Explanations for Multi-Agent Reinforcement Learning","date":"2022-04-26","arxiv_id":"2204.12568","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-bid-long-term-multi-agent","slug":"learning-to-bid-long-term-multi-agent","title":"Learning to Bid Long-Term: Multi-Agent Reinforcement Learning with Long-Term and Sparse Reward in Repeated Auction Games","date":"2022-04-05","arxiv_id":"2204.02268","repositories_listed":1,"syntology":null},{"url":"/paper/quantum-multi-agent-reinforcement-learning","slug":"quantum-multi-agent-reinforcement-learning","title":"Quantum Multi-Agent Reinforcement Learning via Variational Quantum Circuit Design","date":"2022-03-20","arxiv_id":"2203.10443","repositories_listed":1,"syntology":null},{"url":"/paper/coach-assisted-multi-agent-reinforcement","slug":"coach-assisted-multi-agent-reinforcement","title":"Coach-assisted Multi-Agent Reinforcement Learning Framework for Unexpected Crashed Agents","date":"2022-03-16","arxiv_id":"2203.08454","repositories_listed":1,"syntology":null},{"url":"/paper/ctds-centralized-teacher-with-decentralized","slug":"ctds-centralized-teacher-with-decentralized","title":"CTDS: Centralized Teacher with Decentralized Student for Multi-Agent Reinforcement Learning","date":"2022-03-16","arxiv_id":"2203.08412","repositories_listed":1,"syntology":null},{"url":"/paper/pmic-improving-multi-agent-reinforcement-1","slug":"pmic-improving-multi-agent-reinforcement-1","title":"PMIC: Improving Multi-Agent Reinforcement Learning with Progressive Mutual Information Collaboration","date":"2022-03-16","arxiv_id":"2203.08553","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pmic-improving-multi-agent-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2203.08553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08553"}},"official":{"repos":["yeshenpy/pmic"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reliably-re-acting-to-partner-s-actions-with","slug":"reliably-re-acting-to-partner-s-actions-with","title":"Reliably Re-Acting to Partner's Actions with the Social Intrinsic Motivation of Transfer Empowerment","date":"2022-03-07","arxiv_id":"2203.03355","repositories_listed":1,"syntology":null},{"url":"/paper/can-mean-field-control-mfc-approximate","slug":"can-mean-field-control-mfc-approximate","title":"Can Mean Field Control (MFC) Approximate Cooperative Multi Agent Reinforcement Learning (MARL) with Non-Uniform Interaction?","date":"2022-02-28","arxiv_id":"2203.00035","repositories_listed":1,"syntology":null},{"url":"/paper/coordinate-aligned-multi-camera-collaboration","slug":"coordinate-aligned-multi-camera-collaboration","title":"Coordinate-Aligned Multi-Camera Collaboration for Active Multi-Object Tracking","date":"2022-02-22","arxiv_id":"2202.10881","repositories_listed":1,"syntology":null},{"url":"/paper/a-multi-agent-reinforcement-learning","slug":"a-multi-agent-reinforcement-learning","title":"A Multi-Agent Reinforcement Learning Framework for Off-Policy Evaluation in Two-sided Markets","date":"2022-02-21","arxiv_id":"2202.10574","repositories_listed":1,"syntology":null},{"url":"/paper/dqmix-a-distributional-perspective-on-multi","slug":"dqmix-a-distributional-perspective-on-multi","title":"MCMARL: Parameterizing Value Function via Mixture of Categorical Distributions for Multi-Agent Reinforcement Learning","date":"2022-02-21","arxiv_id":"2202.10134","repositories_listed":1,"syntology":null},{"url":"/paper/cooperative-artificial-intelligence","slug":"cooperative-artificial-intelligence","title":"Cooperative Artificial Intelligence","date":"2022-02-20","arxiv_id":"2202.09859","repositories_listed":1,"syntology":null},{"url":"/paper/shaping-advice-in-deep-reinforcement-learning","slug":"shaping-advice-in-deep-reinforcement-learning","title":"Shaping Advice in Deep Reinforcement Learning","date":"2022-02-19","arxiv_id":"2202.09489","repositories_listed":1,"syntology":null},{"url":"/paper/darl1n-distributed-multi-agent-reinforcement","slug":"darl1n-distributed-multi-agent-reinforcement","title":"Distributed Multi-Agent Reinforcement Learning with One-hop Neighbors and Compute Straggler Mitigation","date":"2022-02-18","arxiv_id":"2202.09019","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-path-finding-with-prioritized","slug":"multi-agent-path-finding-with-prioritized","title":"Multi-Agent Path Finding with Prioritized Communication Learning","date":"2022-02-08","arxiv_id":"2202.03634","repositories_listed":1,"syntology":null},{"url":"/paper/iterated-reasoning-with-mutual-information-in-1","slug":"iterated-reasoning-with-mutual-information-in-1","title":"Iterated Reasoning with Mutual Information in Cooperative and Byzantine Decentralized Teaming","date":"2022-01-20","arxiv_id":"2201.08484","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":1,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/iterated-reasoning-with-mutual-information-in-1#ran","syntology_url":"https://syntology.ai/paper/2201.08484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.08484"}},"official":{"repos":["core-robotics-lab/infopg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/modeling-bounded-rationality-in-multi-agent-1","slug":"modeling-bounded-rationality-in-multi-agent-1","title":"Solving Dynamic Principal-Agent Problems with a Rationally Inattentive Principal","date":"2022-01-18","arxiv_id":"2202.01691","repositories_listed":1,"syntology":null},{"url":"/paper/agent-temporal-attention-for-reward","slug":"agent-temporal-attention-for-reward","title":"Agent-Temporal Attention for Reward Redistribution in Episodic Multi-Agent Reinforcement Learning","date":"2022-01-12","arxiv_id":"2201.04612","repositories_listed":1,"syntology":null},{"url":"/paper/value-function-factorisation-with-hypergraph","slug":"value-function-factorisation-with-hypergraph","title":"Cooperative Multi-Agent Reinforcement Learning with Hypergraph Convolution","date":"2021-12-09","arxiv_id":"2112.06771","repositories_listed":1,"syntology":null},{"url":"/paper/self-organized-polynomial-time-coordination-1","slug":"self-organized-polynomial-time-coordination-1","title":"Self-Organized Polynomial-Time Coordination Graphs","date":"2021-12-07","arxiv_id":"2112.03547","repositories_listed":1,"syntology":null},{"url":"/paper/mdpgt-momentum-based-decentralized-policy","slug":"mdpgt-momentum-based-decentralized-policy","title":"MDPGT: Momentum-based Decentralized Policy Gradient Tracking","date":"2021-12-06","arxiv_id":"2112.02813","repositories_listed":1,"syntology":null},{"url":"/paper/offline-pre-trained-multi-agent-decision-1","slug":"offline-pre-trained-multi-agent-decision-1","title":"Offline Pre-trained Multi-Agent Decision Transformer: One Big Sequence Model Tackles All SMAC Tasks","date":"2021-12-06","arxiv_id":"2112.02845","repositories_listed":1,"syntology":null},{"url":"/paper/neural-auto-curricula-in-two-player-zero-sum","slug":"neural-auto-curricula-in-two-player-zero-sum","title":"Neural Auto-Curricula in Two-Player Zero-Sum Games","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/pessimism-meets-invariance-provably-efficient","slug":"pessimism-meets-invariance-provably-efficient","title":"Pessimism Meets Invariance: Provably Efficient Offline Mean-Field Multi-Agent RL","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/regularized-softmax-deep-multi-agent-q","slug":"regularized-softmax-deep-multi-agent-q","title":"Regularized Softmax Deep Multi-Agent Q-Learning","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/vast-value-function-factorization-with","slug":"vast-value-function-factorization-with","title":"VAST: Value Function Factorization with Variable Agent Sub-Teams","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-lingual-agents-through-multi-headed","slug":"multi-lingual-agents-through-multi-headed","title":"Multi-lingual agents through multi-headed neural networks","date":"2021-11-22","arxiv_id":"2111.11129","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-lingual-agents-through-multi-headed#ran","syntology_url":"https://syntology.ai/paper/2111.11129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.11129"}},"official":{"repos":["jon17591/multi-lingual-agents"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/off-policy-correction-for-multi-agent","slug":"off-policy-correction-for-multi-agent","title":"Off-Policy Correction For Multi-Agent Reinforcement Learning","date":"2021-11-22","arxiv_id":"2111.11229","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/off-policy-correction-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2111.11229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.11229"}},"official":{"repos":["awarelab/seed_rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/plan-better-amid-conservatism-offline-multi-1","slug":"plan-better-amid-conservatism-offline-multi-1","title":"Plan Better Amid Conservatism: Offline Multi-Agent Reinforcement Learning with Actor Rectification","date":"2021-11-22","arxiv_id":"2111.11188","repositories_listed":1,"syntology":null},{"url":"/paper/cooperative-multi-agent-reinforcement-4","slug":"cooperative-multi-agent-reinforcement-4","title":"Cooperative multi-agent reinforcement learning for high-dimensional nonequilibrium control","date":"2021-11-12","arxiv_id":"2111.06875","repositories_listed":1,"syntology":null},{"url":"/paper/resilient-consensus-based-multi-agent","slug":"resilient-consensus-based-multi-agent","title":"Resilient Consensus-based Multi-agent Reinforcement Learning with Function Approximation","date":"2021-11-12","arxiv_id":"2111.06776","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-use-and-misuse-of-absorbing-states-in","slug":"on-the-use-and-misuse-of-absorbing-states-in","title":"On the Use and Misuse of Absorbing States in Multi-agent Reinforcement Learning","date":"2021-11-10","arxiv_id":"2111.05992","repositories_listed":1,"syntology":null},{"url":"/paper/powergridworld-a-framework-for-multi-agent","slug":"powergridworld-a-framework-for-multi-agent","title":"PowerGridworld: A Framework for Multi-Agent Reinforcement Learning in Power Systems","date":"2021-11-10","arxiv_id":"2111.05969","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/powergridworld-a-framework-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2111.05969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.05969"}},"official":{"repos":["nrel/powergridworld"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/variational-automatic-curriculum-learning-for","slug":"variational-automatic-curriculum-learning-for","title":"Variational Automatic Curriculum Learning for Sparse-Reward Cooperative Multi-Agent Problems","date":"2021-11-08","arxiv_id":"2111.04613","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/variational-automatic-curriculum-learning-for#ran","syntology_url":"https://syntology.ai/paper/2111.04613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.04613"}},"official":null}},{"url":"/paper/cross-modality-3d-navigation-using","slug":"cross-modality-3d-navigation-using","title":"Cross Modality 3D Navigation Using Reinforcement Learning and Neural Style Transfer","date":"2021-11-05","arxiv_id":"2111.03485","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-active","slug":"multi-agent-reinforcement-learning-for-active","title":"Multi-Agent Reinforcement Learning for Active Voltage Control on Power Distribution Networks","date":"2021-10-27","arxiv_id":"2110.14300","repositories_listed":1,"syntology":null}],"record_sha256":"b9f2406088e4ecc8ceedda5c2cc1c6278aeea9e20d527b86ffd4ca4999495ae1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}