{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multi-agent-reinforcement-learning/papers/5","list_of":"/task/multi-agent-reinforcement-learning","task":"Multi-agent Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":18,"rows_per_page":100,"rows":[401,500],"of":1718,"counts":{"archive_papers_tagged":1718,"with_a_code_link":522,"where_syntology_ran_a_sample":135,"not_listed_spam_title":0,"listed":1718,"listed_where_code_ran":135,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":119,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":119,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multi-agent-reinforcement-learning","prev":"/task/multi-agent-reinforcement-learning/papers/4","next":"/task/multi-agent-reinforcement-learning/papers/6","papers":[{"url":"/paper/learning-to-simulate-self-driven-particles","slug":"learning-to-simulate-self-driven-particles","title":"Learning to Simulate Self-Driven Particles System with Coordinated Policy Optimization","date":"2021-10-26","arxiv_id":"2110.13827","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/learning-to-simulate-self-driven-particles#ran","syntology_url":"https://syntology.ai/paper/2110.13827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.13827"}},"official":{"repos":["decisionforce/CoPO"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-advisor-q-learning","slug":"multi-agent-advisor-q-learning","title":"Multi-Agent Advisor Q-Learning","date":"2021-10-26","arxiv_id":"2111.00345","repositories_listed":1,"syntology":null},{"url":"/paper/ace-hgnn-adaptive-curvature-exploration","slug":"ace-hgnn-adaptive-curvature-exploration","title":"ACE-HGNN: Adaptive Curvature Exploration Hyperbolic Graph Neural Network","date":"2021-10-15","arxiv_id":"2110.07888","repositories_listed":1,"syntology":null},{"url":"/paper/homogeneous-learning-self-attention-1","slug":"homogeneous-learning-self-attention-1","title":"Homogeneous Learning: Self-Attention Decentralized Deep Learning","date":"2021-10-11","arxiv_id":"2110.05290","repositories_listed":1,"syntology":null},{"url":"/paper/is-machine-learning-ready-for-traffic","slug":"is-machine-learning-ready-for-traffic","title":"Is Machine Learning Ready for Traffic Engineering Optimization?","date":"2021-09-03","arxiv_id":"2109.01445","repositories_listed":1,"syntology":null},{"url":"/paper/a-scalable-federated-multi-agent-architecture","slug":"a-scalable-federated-multi-agent-architecture","title":"Scalable Multi-agent Reinforcement Learning Algorithm for Wireless Networks","date":"2021-08-01","arxiv_id":"2108.00506","repositories_listed":1,"syntology":null},{"url":"/paper/strategically-efficient-exploration-in","slug":"strategically-efficient-exploration-in","title":"Strategically Efficient Exploration in Competitive Multi-agent Reinforcement Learning","date":"2021-07-30","arxiv_id":"2107.14698","repositories_listed":1,"syntology":null},{"url":"/paper/a-sustainable-ecosystem-through-emergent","slug":"a-sustainable-ecosystem-through-emergent","title":"A Sustainable Ecosystem through Emergent Cooperation in Multi-Agent Reinforcement Learning","date":"2021-07-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/implicit-communication-as-minimum-entropy","slug":"implicit-communication-as-minimum-entropy","title":"Communicating via Markov Decision Processes","date":"2021-07-17","arxiv_id":"2107.08295","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/implicit-communication-as-minimum-entropy#ran","syntology_url":"https://syntology.ai/paper/2107.08295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.08295"}},"official":{"repos":["schroederdewitt/meme"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mava-a-research-framework-for-distributed","slug":"mava-a-research-framework-for-distributed","title":"Mava: a research library for distributed multi-agent reinforcement learning in JAX","date":"2021-07-03","arxiv_id":"2107.01460","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-visual-navigation","slug":"collaborative-visual-navigation","title":"Collaborative Visual Navigation","date":"2021-07-02","arxiv_id":"2107.01151","repositories_listed":1,"syntology":null},{"url":"/paper/a-game-theoretic-approach-to-multi-agent","slug":"a-game-theoretic-approach-to-multi-agent","title":"A Game-Theoretic Approach to Multi-Agent Trust Region Optimization","date":"2021-06-12","arxiv_id":"2106.06828","repositories_listed":1,"syntology":null},{"url":"/paper/a-cooperative-competitive-multi-agent","slug":"a-cooperative-competitive-multi-agent","title":"A Cooperative-Competitive Multi-Agent Framework for Auto-bidding in Online Advertising","date":"2021-06-11","arxiv_id":"2106.06224","repositories_listed":1,"syntology":null},{"url":"/paper/a-new-formalism-method-and-open-issues-for","slug":"a-new-formalism-method-and-open-issues-for","title":"A New Formalism, Method and Open Issues for Zero-Shot Coordination","date":"2021-06-11","arxiv_id":"2106.06613","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-new-formalism-method-and-open-issues-for#ran","syntology_url":"https://syntology.ai/paper/2106.06613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06613"}},"official":{"repos":["johannestreutlein/op-tie-breaking"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/douzero-mastering-doudizhu-with-self-play","slug":"douzero-mastering-doudizhu-with-self-play","title":"DouZero: Mastering DouDizhu with Self-Play Deep Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06135","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/douzero-mastering-doudizhu-with-self-play#ran","syntology_url":"https://syntology.ai/paper/2106.06135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06135"}},"official":{"repos":["kwai/DouZero"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/ermas-becoming-robust-to-reward-function-sim","slug":"ermas-becoming-robust-to-reward-function-sim","title":"Learning to Play General-Sum Games Against Multiple Boundedly Rational Agents","date":"2021-06-10","arxiv_id":"2106.05492","repositories_listed":1,"syntology":null},{"url":"/paper/believe-what-you-see-implicit-constraint","slug":"believe-what-you-see-implicit-constraint","title":"Believe What You See: Implicit Constraint Approach for Offline Multi-Agent Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03400","repositories_listed":1,"syntology":null},{"url":"/paper/malib-a-parallel-framework-for-population","slug":"malib-a-parallel-framework-for-population","title":"MALib: A Parallel Framework for Population-based Multi-agent Reinforcement Learning","date":"2021-06-05","arxiv_id":"2106.07551","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/malib-a-parallel-framework-for-population#ran","syntology_url":"https://syntology.ai/paper/2106.07551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.07551"}},"official":{"repos":["sjtu-marl/malib"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/discovering-multi-agent-auto-curricula-in-two","slug":"discovering-multi-agent-auto-curricula-in-two","title":"Neural Auto-Curricula","date":"2021-06-04","arxiv_id":"2106.02745","repositories_listed":1,"syntology":null},{"url":"/paper/shaq-incorporating-shapley-value-theory-into","slug":"shaq-incorporating-shapley-value-theory-into","title":"SHAQ: Incorporating Shapley Value Theory into Multi-Agent Q-Learning","date":"2021-05-31","arxiv_id":"2105.15013","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/shaq-incorporating-shapley-value-theory-into#ran","syntology_url":"https://syntology.ai/paper/2105.15013","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.15013"}},"official":{"repos":["hsvgbkhgbv/shapley-q-learning"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/from-motor-control-to-team-play-in-simulated","slug":"from-motor-control-to-team-play-in-simulated","title":"From Motor Control to Team Play in Simulated Humanoid Football","date":"2021-05-25","arxiv_id":"2105.12196","repositories_listed":1,"syntology":null},{"url":"/paper/cooperative-multi-agent-reinforcement-5","slug":"cooperative-multi-agent-reinforcement-5","title":"Cooperative Multi-Agent Reinforcement Learning with Sequential Credit Assignment","date":"2021-05-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/coach-player-multi-agent-reinforcement","slug":"coach-player-multi-agent-reinforcement","title":"Coach-Player Multi-Agent Reinforcement Learning for Dynamic Team Composition","date":"2021-05-18","arxiv_id":"2105.08692","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coach-player-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2105.08692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.08692"}},"official":{"repos":["cranial-xix/marl-copa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-multi-agent-policy-optimization","slug":"model-based-multi-agent-policy-optimization","title":"Model-based Multi-agent Policy Optimization with Adaptive Opponent-wise Rollouts","date":"2021-05-07","arxiv_id":"2105.03363","repositories_listed":1,"syntology":null},{"url":"/paper/decomposed-soft-actor-critic-method-for","slug":"decomposed-soft-actor-critic-method-for","title":"Decomposed Soft Actor-Critic Method for Cooperative Multi-Agent Reinforcement Learning","date":"2021-04-14","arxiv_id":"2104.06655","repositories_listed":1,"syntology":null},{"url":"/paper/a-coevolutionairy-approach-to-deep-multi","slug":"a-coevolutionairy-approach-to-deep-multi","title":"A coevolutionary approach to deep multi-agent reinforcement learning","date":"2021-04-12","arxiv_id":"2104.05610","repositories_listed":1,"syntology":null},{"url":"/paper/c-coma-a-continual-reinforcement-learning","slug":"c-coma-a-continual-reinforcement-learning","title":"C-COMA: A CONTINUAL REINFORCEMENT LEARNING MODEL FOR DYNAMIC MULTIAGENT ENVIRONMENTS","date":"2021-04-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/shaping-advice-in-deep-multi-agent","slug":"shaping-advice-in-deep-multi-agent","title":"Shaping Advice in Deep Multi-Agent Reinforcement Learning","date":"2021-03-29","arxiv_id":"2103.15941","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-explanation-with-multi-agent","slug":"counterfactual-explanation-with-multi-agent","title":"Counterfactual Explanation with Multi-Agent Reinforcement Learning for Drug Target Prediction","date":"2021-03-24","arxiv_id":"2103.12983","repositories_listed":1,"syntology":null},{"url":"/paper/the-ai-arena-a-framework-for-distributed","slug":"the-ai-arena-a-framework-for-distributed","title":"The AI Arena: A Framework for Distributed Multi-Agent Reinforcement Learning","date":"2021-03-09","arxiv_id":"2103.05737","repositories_listed":1,"syntology":null},{"url":"/paper/deepfreight-a-model-free-deep-reinforcement","slug":"deepfreight-a-model-free-deep-reinforcement","title":"DeepFreight: Integrating Deep Reinforcement Learning and Mixed Integer Programming for Multi-transfer Truck Freight Delivery","date":"2021-03-05","arxiv_id":"2103.03450","repositories_listed":1,"syntology":null},{"url":"/paper/balancing-rational-and-other-regarding","slug":"balancing-rational-and-other-regarding","title":"Balancing Rational and Other-Regarding Preferences in Cooperative-Competitive Environments","date":"2021-02-24","arxiv_id":"2102.12307","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-of-3d","slug":"multi-agent-reinforcement-learning-of-3d","title":"Multi-Agent Reinforcement Learning of 3D Furniture Layout Simulation in Indoor Graphics Scenes","date":"2021-02-18","arxiv_id":"2102.09137","repositories_listed":1,"syntology":null},{"url":"/paper/dfac-framework-factorizing-the-value-function","slug":"dfac-framework-factorizing-the-value-function","title":"DFAC Framework: Factorizing the Value Function via Quantile Mixture for Multi-Agent Distributional Q-Learning","date":"2021-02-16","arxiv_id":"2102.07936","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/dfac-framework-factorizing-the-value-function#ran","syntology_url":"https://syntology.ai/paper/2102.07936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.07936"}},"official":{"repos":["j3soon/dfac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/intelligent-electric-vehicle-charging","slug":"intelligent-electric-vehicle-charging","title":"Intelligent Electric Vehicle Charging Recommendation Based on Multi-Agent Reinforcement Learning","date":"2021-02-15","arxiv_id":"2102.07359","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-multi-agent-reinforcement-learning","slug":"scaling-multi-agent-reinforcement-learning","title":"Scaling Multi-Agent Reinforcement Learning with Selective Parameter Sharing","date":"2021-02-15","arxiv_id":"2102.07475","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/scaling-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2102.07475","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.07475"}},"official":{"repos":["uoe-agents/seps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-with","slug":"multi-agent-reinforcement-learning-with","title":"Multi-Agent Reinforcement Learning with Temporal Logic Specifications","date":"2021-02-01","arxiv_id":"2102.00582","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2102.00582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.00582"}},"official":{"repos":["lrhammond/almanac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/data-sharing-games","slug":"data-sharing-games","title":"Data sharing games","date":"2021-01-26","arxiv_id":"2101.10721","repositories_listed":1,"syntology":null},{"url":"/paper/updet-universal-multi-agent-reinforcement","slug":"updet-universal-multi-agent-reinforcement","title":"UPDeT: Universal Multi-agent Reinforcement Learning via Policy Decoupling with Transformers","date":"2021-01-20","arxiv_id":"2101.08001","repositories_listed":1,"syntology":null},{"url":"/paper/hammer-multi-level-coordination-of","slug":"hammer-multi-level-coordination-of","title":"HAMMER: Multi-Level Coordination of Reinforcement Learning Agents via Learned Messaging","date":"2021-01-18","arxiv_id":"2102.00824","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-trust-region-learning","slug":"multi-agent-trust-region-learning","title":"Multi-Agent Trust Region Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-7","slug":"multi-agent-reinforcement-learning-for-7","title":"Multi-Agent Reinforcement Learning for Unmanned Aerial Vehicle Coordination by Multi-Critic Policy Gradient Optimization","date":"2020-12-31","arxiv_id":"2012.15472","repositories_listed":1,"syntology":null},{"url":"/paper/cooperative-policy-learning-with-pre-trained","slug":"cooperative-policy-learning-with-pre-trained","title":"Cooperative Policy Learning with Pre-trained Heterogeneous Observation Representations","date":"2020-12-24","arxiv_id":"2012.13099","repositories_listed":1,"syntology":null},{"url":"/paper/qvmix-and-qvmix-max-extending-the-deep","slug":"qvmix-and-qvmix-max-extending-the-deep","title":"QVMix and QVMix-Max: Extending the Deep Quality-Value Family of Algorithms to Cooperative Multi-Agent Reinforcement Learning","date":"2020-12-22","arxiv_id":"2012.12062","repositories_listed":1,"syntology":null},{"url":"/paper/citylearn-standardizing-research-in-multi","slug":"citylearn-standardizing-research-in-multi","title":"CityLearn: Standardizing Research in Multi-Agent Reinforcement Learning for Demand Response and Urban Energy Management","date":"2020-12-18","arxiv_id":"2012.10504","repositories_listed":1,"syntology":null},{"url":"/paper/learning-multi-agent-communication-through","slug":"learning-multi-agent-communication-through","title":"Learning Multi-Agent Communication through Structured Attentive Reasoning","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/tleague-a-framework-for-competitive-self-play","slug":"tleague-a-framework-for-competitive-self-play","title":"TLeague: A Framework for Competitive Self-Play based Distributed Multi-Agent Reinforcement Learning","date":"2020-11-25","arxiv_id":"2011.12895","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-reinforcement-learning-policies-for","slug":"scalable-reinforcement-learning-policies-for","title":"Scalable Reinforcement Learning Policies for Multi-Agent Control","date":"2020-11-16","arxiv_id":"2011.08055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-reinforcement-learning-policies-for#ran","syntology_url":"https://syntology.ai/paper/2011.08055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.08055"}},"official":{"repos":["christopher-hsu/scalableMARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimizing-large-scale-fleet-management-on-a","slug":"optimizing-large-scale-fleet-management-on-a","title":"Optimizing Large-Scale Fleet Management on a Road Network using Multi-Agent Deep Reinforcement Learning with Graph Neural Network","date":"2020-11-12","arxiv_id":"2011.06175","repositories_listed":1,"syntology":null},{"url":"/paper/emergent-reciprocity-and-team-formation-from","slug":"emergent-reciprocity-and-team-formation-from","title":"Emergent Reciprocity and Team Formation from Randomized Uncertain Social Preferences","date":"2020-11-10","arxiv_id":"2011.05373","repositories_listed":1,"syntology":null},{"url":"/paper/learning-a-decentralized-multi-arm-motion","slug":"learning-a-decentralized-multi-arm-motion","title":"Learning a Decentralized Multi-arm Motion Planner","date":"2020-11-05","arxiv_id":"2011.02608","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for","slug":"multi-agent-reinforcement-learning-for","title":"Multi-Agent Reinforcement Learning for Visibility-based Persistent Monitoring","date":"2020-11-02","arxiv_id":"2011.01129","repositories_listed":1,"syntology":null},{"url":"/paper/an-overview-of-multi-agent-reinforcement","slug":"an-overview-of-multi-agent-reinforcement","title":"Game-Theoretic Multiagent Reinforcement Learning","date":"2020-11-01","arxiv_id":"2011.00583","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-overview-of-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2011.00583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00583"}},"official":null}},{"url":"/paper/succinct-and-robust-multi-agent-communication","slug":"succinct-and-robust-multi-agent-communication","title":"Succinct and Robust Multi-Agent Communication With Temporal Message Control","date":"2020-10-27","arxiv_id":"2010.14391","repositories_listed":1,"syntology":null},{"url":"/paper/multi-uav-path-planning-for-wireless-data","slug":"multi-uav-path-planning-for-wireless-data","title":"Multi-UAV Path Planning for Wireless Data Harvesting with Deep Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12461","repositories_listed":1,"syntology":null},{"url":"/paper/a-game-theoretic-analysis-of-networked-system","slug":"a-game-theoretic-analysis-of-networked-system","title":"A game-theoretic analysis of networked system control for common-pool resource management using multi-agent reinforcement learning","date":"2020-10-15","arxiv_id":"2010.07777","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-game-theoretic-analysis-of-networked-system#ran","syntology_url":"https://syntology.ai/paper/2010.07777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07777"}},"official":{"repos":["instadeepai/EGTA-NMARL"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-trust-region-policy-optimization","slug":"multi-agent-trust-region-policy-optimization","title":"Multi-Agent Trust Region Policy Optimization","date":"2020-10-15","arxiv_id":"2010.07916","repositories_listed":1,"syntology":{"n":18,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":7,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-trust-region-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2010.07916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07916"}},"official":null}},{"url":"/paper/distributed-resource-allocation-with-multi","slug":"distributed-resource-allocation-with-multi","title":"Distributed Resource Allocation with Multi-Agent Deep Reinforcement Learning for 5G-V2V Communication","date":"2020-10-11","arxiv_id":"2010.05290","repositories_listed":1,"syntology":null},{"url":"/paper/graph-convolutional-value-decomposition-in-1","slug":"graph-convolutional-value-decomposition-in-1","title":"Graph Convolutional Value Decomposition in Multi-Agent Reinforcement Learning","date":"2020-10-09","arxiv_id":"2010.04740","repositories_listed":1,"syntology":null},{"url":"/paper/energy-based-surprise-minimization-for-multi","slug":"energy-based-surprise-minimization-for-multi","title":"Energy-based Surprise Minimization for Multi-Agent Value Factorization","date":"2020-09-16","arxiv_id":"2009.09842","repositories_listed":1,"syntology":null},{"url":"/paper/optimization-of-operation-parameters-towards","slug":"optimization-of-operation-parameters-towards","title":"Optimal control towards sustainable wastewater treatment plants based on multi-agent reinforcement learning","date":"2020-08-19","arxiv_id":"2008.10417","repositories_listed":1,"syntology":null},{"url":"/paper/communicative-reinforcement-learning-agents","slug":"communicative-reinforcement-learning-agents","title":"Communicative Reinforcement Learning Agents for Landmark Detection in Brain Images","date":"2020-08-18","arxiv_id":"2008.08055","repositories_listed":1,"syntology":null},{"url":"/paper/towards-closing-the-sim-to-real-gap-in","slug":"towards-closing-the-sim-to-real-gap-in","title":"Towards Closing the Sim-to-Real Gap in Collaborative Multi-Robot Deep Reinforcement Learning","date":"2020-08-18","arxiv_id":"2008.07875","repositories_listed":1,"syntology":null},{"url":"/paper/the-emergence-of-adversarial-communication-in","slug":"the-emergence-of-adversarial-communication-in","title":"The Emergence of Adversarial Communication in Multi-Agent Reinforcement Learning","date":"2020-08-06","arxiv_id":"2008.02616","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-emergence-of-adversarial-communication-in#ran","syntology_url":"https://syntology.ai/paper/2008.02616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.02616"}},"official":{"repos":["proroklab/adversarial_comms"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-step-reinforcement-learning-for-single","slug":"multi-step-reinforcement-learning-for-single","title":"Multi-Step Reinforcement Learning for Single Image Super-Resolution","date":"2020-07-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-multi-agent-decomposed-policy","slug":"off-policy-multi-agent-decomposed-policy","title":"Off-Policy Multi-Agent Decomposed Policy Gradients","date":"2020-07-24","arxiv_id":"2007.12322","repositories_listed":1,"syntology":null},{"url":"/paper/value-decomposition-multi-agent-actor-critics","slug":"value-decomposition-multi-agent-actor-critics","title":"Value-Decomposition Multi-Agent Actor-Critics","date":"2020-07-24","arxiv_id":"2007.12306","repositories_listed":1,"syntology":null},{"url":"/paper/battlesnake-challenge-a-multi-agent","slug":"battlesnake-challenge-a-multi-agent","title":"Battlesnake Challenge: A Multi-agent Reinforcement Learning Playground with Human-in-the-loop","date":"2020-07-20","arxiv_id":"2007.10504","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-communication-learning-in","slug":"reinforcement-communication-learning-in","title":"Reinforcement Communication Learning in Different Social Network Structures","date":"2020-07-19","arxiv_id":"2007.09820","repositories_listed":1,"syntology":null},{"url":"/paper/curriculum-learning-for-multilevel-budgeted","slug":"curriculum-learning-for-multilevel-budgeted","title":"Curriculum learning for multilevel budgeted combinatorial problems","date":"2020-07-07","arxiv_id":"2007.03151","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/curriculum-learning-for-multilevel-budgeted#ran","syntology_url":"https://syntology.ai/paper/2007.03151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.03151"}},"official":{"repos":["AdelNabli/MCN"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-implicit-credit-assignment-for-multi","slug":"learning-implicit-credit-assignment-for-multi","title":"Learning Implicit Credit Assignment for Cooperative Multi-Agent Reinforcement Learning","date":"2020-07-06","arxiv_id":"2007.02529","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/learning-implicit-credit-assignment-for-multi#ran","syntology_url":"https://syntology.ai/paper/2007.02529","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02529"}},"official":{"repos":["mzho7212/LICA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/deep-implicit-coordination-graphs-for-multi","slug":"deep-implicit-coordination-graphs-for-multi","title":"Deep Implicit Coordination Graphs for Multi-agent Reinforcement Learning","date":"2020-06-19","arxiv_id":"2006.11438","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-ridesharing-dispatch-using-multi","slug":"efficient-ridesharing-dispatch-using-multi","title":"Efficient Ridesharing Dispatch Using Multi-Agent Reinforcement Learning","date":"2020-06-18","arxiv_id":"2006.10897","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-reinforcement-learning-in-multi","slug":"distributed-reinforcement-learning-in-multi","title":"Multi-Agent Reinforcement Learning in Stochastic Networked Systems","date":"2020-06-11","arxiv_id":"2006.06555","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/distributed-reinforcement-learning-in-multi#ran","syntology_url":"https://syntology.ai/paper/2006.06555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06555"}},"official":null}},{"url":"/paper/learning-individually-inferred-communication","slug":"learning-individually-inferred-communication","title":"Learning Individually Inferred Communication for Multi-Agent Cooperation","date":"2020-06-11","arxiv_id":"2006.06455","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-model-opponent-learning","slug":"learning-to-model-opponent-learning","title":"Learning to Model Opponent Learning","date":"2020-06-06","arxiv_id":"2006.03923","repositories_listed":1,"syntology":null},{"url":"/paper/parameter-sharing-is-surprisingly-useful-for","slug":"parameter-sharing-is-surprisingly-useful-for","title":"Revisiting Parameter Sharing in Multi-Agent Deep Reinforcement Learning","date":"2020-05-27","arxiv_id":"2005.13625","repositories_listed":1,"syntology":null},{"url":"/paper/delay-aware-multi-agent-reinforcement","slug":"delay-aware-multi-agent-reinforcement","title":"Delay-Aware Multi-Agent Reinforcement Learning for Cooperative and Competitive Environments","date":"2020-05-11","arxiv_id":"2005.05441","repositories_listed":1,"syntology":null},{"url":"/paper/gifting-in-multi-agent-reinforcement-learning","slug":"gifting-in-multi-agent-reinforcement-learning","title":"Gifting in multi-agent reinforcement learning","date":"2020-05-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/marleme-a-multi-agent-reinforcement-learning","slug":"marleme-a-multi-agent-reinforcement-learning","title":"MARLeME: A Multi-Agent Reinforcement Learning Model Extraction Library","date":"2020-04-16","arxiv_id":"2004.07928","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-2","slug":"multi-agent-reinforcement-learning-for-2","title":"Multi-agent Reinforcement Learning for Networked System Control","date":"2020-04-03","arxiv_id":"2004.01339","repositories_listed":1,"syntology":{"n":16,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":14,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":16,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-2#ran","syntology_url":"https://syntology.ai/paper/2004.01339","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.01339"}},"official":{"repos":["cts198859/deeprl_network"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":14,"ran_from_kinds":["official"]}}},{"url":"/paper/information-state-embedding-in-partially","slug":"information-state-embedding-in-partially","title":"Information State Embedding in Partially Observable Cooperative Multi-Agent Reinforcement Learning","date":"2020-04-02","arxiv_id":"2004.01098","repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-population-curriculum-for-1","slug":"evolutionary-population-curriculum-for-1","title":"Evolutionary Population Curriculum for Scaling Multi-Agent Reinforcement Learning","date":"2020-03-23","arxiv_id":"2003.10423","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evolutionary-population-curriculum-for-1#ran","syntology_url":"https://syntology.ai/paper/2003.10423","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.10423"}},"official":{"repos":["qian18long/epciclr2020"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/who2com-collaborative-perception-via","slug":"who2com-collaborative-perception-via","title":"Who2com: Collaborative Perception via Learnable Handshake Communication","date":"2020-03-21","arxiv_id":"2003.09575","repositories_listed":1,"syntology":null},{"url":"/paper/monotonic-value-function-factorisation-for","slug":"monotonic-value-function-factorisation-for","title":"Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.08839","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monotonic-value-function-factorisation-for#ran","syntology_url":"https://syntology.ai/paper/2003.08839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08839"}},"official":{"repos":["oxwhirl/pymarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-robustness-of-cooperative-multi-agent","slug":"on-the-robustness-of-cooperative-multi-agent","title":"On the Robustness of Cooperative Multi-Agent Reinforcement Learning","date":"2020-03-08","arxiv_id":"2003.03722","repositories_listed":1,"syntology":null},{"url":"/paper/ig-rl-inductive-graph-reinforcement-learning","slug":"ig-rl-inductive-graph-reinforcement-learning","title":"IG-RL: Inductive Graph Reinforcement Learning for Massive-Scale Traffic Signal Control","date":"2020-03-06","arxiv_id":"2003.05738","repositories_listed":1,"syntology":null},{"url":"/paper/gamma-reward-a-novel-multi-agent","slug":"gamma-reward-a-novel-multi-agent","title":"Learning Scalable Multi-Agent Coordination by Spatial Differentiation for Traffic Signal Control","date":"2020-02-27","arxiv_id":"2002.11874","repositories_listed":1,"syntology":null},{"url":"/paper/extended-markov-games-to-learn-multiple-tasks","slug":"extended-markov-games-to-learn-multiple-tasks","title":"Extended Markov Games to Learn Multiple Tasks in Multi-Agent Reinforcement Learning","date":"2020-02-14","arxiv_id":"2002.06000","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-cooperative-multi-agent","slug":"hierarchical-cooperative-multi-agent","title":"Hierarchical Cooperative Multi-Agent Reinforcement Learning with Skill Discovery","date":"2019-12-07","arxiv_id":"1912.03558","repositories_listed":1,"syntology":null},{"url":"/paper/liir-learning-individual-intrinsic-reward-in","slug":"liir-learning-individual-intrinsic-reward-in","title":"LIIR: Learning Individual Intrinsic Reward in Multi-Agent Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/smix-enhancing-centralized-value-functions","slug":"smix-enhancing-centralized-value-functions","title":"SMIX($λ$): Enhancing Centralized Value Functions for Cooperative Multi-Agent Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04094","repositories_listed":1,"syntology":null},{"url":"/paper/convergent-policy-optimization-for-safe","slug":"convergent-policy-optimization-for-safe","title":"Convergent Policy Optimization for Safe Reinforcement Learning","date":"2019-10-26","arxiv_id":"1910.12156","repositories_listed":1,"syntology":null},{"url":"/paper/a-structured-prediction-approach-for","slug":"a-structured-prediction-approach-for","title":"A Structured Prediction Approach for Generalization in Cooperative Multi-Agent Reinforcement Learning","date":"2019-10-19","arxiv_id":"1910.08809","repositories_listed":1,"syntology":null},{"url":"/paper/universal-policies-to-learn-them-all","slug":"universal-policies-to-learn-them-all","title":"Universal Policies to Learn Them All","date":"2019-08-24","arxiv_id":"1908.09184","repositories_listed":1,"syntology":null},{"url":"/paper/health-informed-policy-gradients-for-multi","slug":"health-informed-policy-gradients-for-multi","title":"Health-Informed Policy Gradients for Multi-Agent Reinforcement Learning","date":"2019-08-02","arxiv_id":"1908.01022","repositories_listed":1,"syntology":null},{"url":"/paper/multiple-landmark-detection-using-multi-agent","slug":"multiple-landmark-detection-using-multi-agent","title":"Multiple Landmark Detection using Multi-Agent Reinforcement Learning","date":"2019-06-30","arxiv_id":"1907.00318","repositories_listed":1,"syntology":null},{"url":"/paper/finding-friend-and-foe-in-multi-agent-games","slug":"finding-friend-and-foe-in-multi-agent-games","title":"Finding Friend and Foe in Multi-Agent Games","date":"2019-06-05","arxiv_id":"1906.02330","repositories_listed":1,"syntology":null},{"url":"/paper/coordinated-exploration-via-intrinsic-rewards","slug":"coordinated-exploration-via-intrinsic-rewards","title":"Coordinated Exploration via Intrinsic Rewards for Multi-Agent Reinforcement Learning","date":"2019-05-28","arxiv_id":"1905.12127","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coordinated-exploration-via-intrinsic-rewards#ran","syntology_url":"https://syntology.ai/paper/1905.12127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.12127"}},"official":{"repos":["shariqiqbal2810/Multi-Explore"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-regularized-opponent-model-with-maximum","slug":"a-regularized-opponent-model-with-maximum","title":"A Regularized Opponent Model with Maximum Entropy Objective","date":"2019-05-17","arxiv_id":"1905.08087","repositories_listed":1,"syntology":null}],"record_sha256":"051c0f0dc5b5c60243d3318c72585d9c4f41fc7523e65a1d9a7f88e5247f4feb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}