{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/starcraft-ii/papers/2","list_of":"/task/starcraft-ii","task":"Starcraft II","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,175],"of":175,"counts":{"archive_papers_tagged":175,"with_a_code_link":90,"where_syntology_ran_a_sample":22,"not_listed_spam_title":0,"listed":175,"listed_where_code_ran":22,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":19,"every_run_a_failure_of_syntologys_instrument":3,"listed_with_a_run_with_no_instrument_failure":19,"listed_every_run_a_failure_of_syntologys_instrument":3,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/starcraft-ii","prev":"/task/starcraft-ii","next":null,"papers":[{"url":null,"slug":"aligning-individual-and-collective-objectives","title":"Aligning Individual and Collective Objectives in Multi-Agent Cooperation","date":"2024-02-19","arxiv_id":"2402.12416","repositories_listed":0,"syntology":null},{"url":null,"slug":"coa-gpt-generative-pre-trained-transformers","title":"COA-GPT: Generative Pre-trained Transformers for Accelerated Course of Action Development in Military Operations","date":"2024-02-01","arxiv_id":"2402.01786","repositories_listed":0,"syntology":null},{"url":null,"slug":"bet-explaining-deep-reinforcement-learning","title":"BET: Explaining Deep Reinforcement Learning through The Error-Prone Decisions","date":"2024-01-14","arxiv_id":"2401.07263","repositories_listed":0,"syntology":null},{"url":null,"slug":"starcraftimage-a-dataset-for-prototyping-1","title":"StarCraftImage: A Dataset For Prototyping Spatial Reasoning Methods For Multi-Agent Environments","date":"2024-01-09","arxiv_id":"2401.04290","repositories_listed":0,"syntology":null},{"url":null,"slug":"dcir-dynamic-consistency-intrinsic-reward-for","title":"DCIR: Dynamic Consistency Intrinsic Reward for Multi-Agent Reinforcement Learning","date":"2023-12-10","arxiv_id":"2312.05783","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-exploiter-a-data-efficient-approach","title":"Minimax Exploiter: A Data Efficient Approach for Competitive Self-Play","date":"2023-11-28","arxiv_id":"2311.17190","repositories_listed":0,"syntology":null},{"url":null,"slug":"mir2-towards-provably-robust-multi-agent","title":"Robust Multi-Agent Reinforcement Learning by Mutual Information Regularization","date":"2023-10-15","arxiv_id":"2310.09833","repositories_listed":0,"syntology":null},{"url":null,"slug":"fidelity-induced-interpretable-policy","title":"Fidelity-Induced Interpretable Policy Extraction for Reinforcement Learning","date":"2023-09-12","arxiv_id":"2309.06097","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-world-model-disentanglement-in","title":"Leveraging World Model Disentanglement in Value-Based Multi-Agent Reinforcement Learning","date":"2023-09-08","arxiv_id":"2309.04615","repositories_listed":0,"syntology":null},{"url":null,"slug":"never-explore-repeatedly-in-multi-agent","title":"Never Explore Repeatedly in Multi-Agent Reinforcement Learning","date":"2023-08-19","arxiv_id":"2308.09909","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-multi-agent-reinforcement-learning","title":"Offline Multi-Agent Reinforcement Learning with Coupled Value Factorization","date":"2023-06-15","arxiv_id":"2306.08900","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-value-decomposition-via-unit-wise","title":"Boosting Value Decomposition via Unit-Wise Attentive State Representation for Cooperative Multi-Agent Reinforcement Learning","date":"2023-05-12","arxiv_id":"2305.07182","repositories_listed":0,"syntology":null},{"url":null,"slug":"svde-scalable-value-decomposition-exploration","title":"SVDE: Scalable Value-Decomposition Exploration for Cooperative Multi-Agent Reinforcement Learning","date":"2023-03-16","arxiv_id":"2303.09058","repositories_listed":0,"syntology":null},{"url":null,"slug":"aiir-mix-multi-agent-reinforcement-learning","title":"AIIR-MIX: Multi-Agent Reinforcement Learning Meets Attention Individual Intrinsic Reward Mixing Network","date":"2023-02-19","arxiv_id":"2302.09531","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-learning-for-relative","title":"CURO: Curriculum Learning for Relative Overgeneralization","date":"2022-12-06","arxiv_id":"2212.02733","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-good-trajectories-in-offline","title":"Learning from Good Trajectories in Offline Multi-Agent Reinforcement Learning","date":"2022-11-28","arxiv_id":"2211.15612","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixrts-toward-interpretable-multi-agent","title":"MIXRTs: Toward Interpretable Multi-Agent Reinforcement Learning via Mixing Recurrent Soft Decision Trees","date":"2022-09-15","arxiv_id":"2209.07225","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-centralised-multi-agent-reinforcement","title":"Taming Multi-Agent Reinforcement Learning with Estimator Variance Reduction","date":"2022-09-02","arxiv_id":"2209.01054","repositories_listed":0,"syntology":null},{"url":null,"slug":"forecasting-evolution-of-clusters-in","title":"Forecasting Evolution of Clusters in Game Agents with Hebbian Learning","date":"2022-08-19","arxiv_id":"2209.06904","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-hebbian-learning-on-point-sets","title":"Unsupervised Hebbian Learning on Point Sets in StarCraft II","date":"2022-07-13","arxiv_id":"2207.12323","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-game-theoretical-analysis-for","title":"Evolutionary Game-Theoretical Analysis for General Multiplayer Asymmetric Games","date":"2022-06-22","arxiv_id":"2206.11114","repositories_listed":0,"syntology":null},{"url":null,"slug":"s2rl-do-we-really-need-to-perceive-all-states","title":"S2RL: Do We Really Need to Perceive All States in Deep Multi-Agent Reinforcement Learning?","date":"2022-06-20","arxiv_id":"2206.11054","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-rewards-a-hierarchical-perspective-on","title":"Beyond Rewards: a Hierarchical Perspective on Offline Multiagent Behavioral Analysis","date":"2022-06-17","arxiv_id":"2206.09046","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-beat-multi-agent-reinforcement-learning","title":"Off-Beat Multi-Agent Reinforcement Learning","date":"2022-05-27","arxiv_id":"2205.13718","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-guide-multiple-heterogeneous","title":"Learning to Guide Multiple Heterogeneous Actors from a Single Human Demonstration via Automatic Curriculum Learning in StarCraft II","date":"2022-05-11","arxiv_id":"2205.05784","repositories_listed":0,"syntology":null},{"url":null,"slug":"ldsa-learning-dynamic-subtask-assignment-in","title":"LDSA: Learning Dynamic Subtask Assignment in Cooperative Multi-Agent Reinforcement Learning","date":"2022-05-05","arxiv_id":"2205.02561","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-transfer-role-assignment-across","title":"Learning to Transfer Role Assignment Across Team Sizes","date":"2022-04-17","arxiv_id":"2204.12937","repositories_listed":0,"syntology":null},{"url":null,"slug":"depthwise-convolution-for-multi-agent","title":"Depthwise Convolution for Multi-Agent Communication with Enhanced Mean-Field Approximation","date":"2022-03-06","arxiv_id":"2203.02896","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-advantage-networks-for-cooperative","title":"Local Advantage Networks for Cooperative Multi-Agent Reinforcement Learning","date":"2021-12-23","arxiv_id":"2112.12458","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-communication-with-graph","title":"CGIBNet: Bandwidth-constrained Communication with Graph Information Bottleneck in Multi-Agent Reinforcement Learning","date":"2021-12-20","arxiv_id":"2112.10374","repositories_listed":0,"syntology":null},{"url":null,"slug":"rmix-learning-risk-sensitive-policies","title":"RMIX: Learning Risk-Sensitive Policies forCooperative Reinforcement Learning Agents","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-games-and-simulators-as-a-platform-for","title":"On games and simulators as a platform for development of artificial intelligence for command and control","date":"2021-10-21","arxiv_id":"2110.11305","repositories_listed":0,"syntology":null},{"url":null,"slug":"containerized-distributed-value-based-multi-1","title":"Containerized Distributed Value-Based Multi-Agent Reinforcement Learning","date":"2021-10-15","arxiv_id":"2110.08169","repositories_listed":0,"syntology":null},{"url":null,"slug":"haven-hierarchical-cooperative-multi-agent","title":"HAVEN: Hierarchical Cooperative Multi-Agent Reinforcement Learning with Dual Coordination Mechanism","date":"2021-10-14","arxiv_id":"2110.07246","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-transformers-for-starcraft","title":"Leveraging Transformers for StarCraft Macromanagement Prediction","date":"2021-10-11","arxiv_id":"2110.05343","repositories_listed":0,"syntology":null},{"url":null,"slug":"marnet-backdoor-attacks-against-value","title":"MARNET: Backdoor Attacks against Value-Decomposition Multi-Agent Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"role-diversity-matters-a-study-of-cooperative","title":"Role Diversity Matters: A Study of Cooperative Training Strategies for Multi-Agent RL","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-approach-to-partial-observability-in-games","title":"An Approach to Partial Observability in Games: Learning to Both Act and Observe","date":"2021-08-11","arxiv_id":"2108.05701","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-multi-agent-transfer-learning","title":"Cooperative Multi-Agent Transfer Learning with Level-Adaptive Credit Assignment","date":"2021-06-01","arxiv_id":"2106.00517","repositories_listed":0,"syntology":null},{"url":null,"slug":"shapley-counterfactual-credits-for-multi","title":"Shapley Counterfactual Credits for Multi-Agent Reinforcement Learning","date":"2021-06-01","arxiv_id":"2106.00285","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-using","title":"Multi-Agent Deep Reinforcement Learning using Attentive Graph Neural Architectures for Real-Time Strategy Games","date":"2021-05-21","arxiv_id":"2105.10211","repositories_listed":0,"syntology":null},{"url":null,"slug":"side-i-infer-the-state-i-want-to-learn","title":"SIDE: State Inference for Partially Observable Cooperative Multi-Agent Reinforcement Learning","date":"2021-05-13","arxiv_id":"2105.06228","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-convolution-for-irregularly-sampled-1","title":"Deep Convolution for Irregularly Sampled Temporal Point Clouds","date":"2021-05-01","arxiv_id":"2105.00137","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-monotonic-value-function-factorization","title":"NQMIX: Non-monotonic Value Function Factorization for Deep Multi-Agent Reinforcement Learning","date":"2021-04-05","arxiv_id":"2104.01939","repositories_listed":0,"syntology":null},{"url":null,"slug":"softmax-with-regularization-better-value","title":"Regularized Softmax Deep Multi-Agent $Q$-Learning","date":"2021-03-22","arxiv_id":"2103.11883","repositories_listed":0,"syntology":null},{"url":null,"slug":"credit-assignment-with-meta-policy-gradient","title":"Credit Assignment with Meta-Policy Gradient for Multi-Agent Reinforcement Learning","date":"2021-02-24","arxiv_id":"2102.12957","repositories_listed":0,"syntology":null},{"url":null,"slug":"rmix-learning-risk-sensitive-policies-for","title":"RMIX: Learning Risk-Sensitive Policies for Cooperative Reinforcement Learning Agents","date":"2021-02-16","arxiv_id":"2102.08159","repositories_listed":0,"syntology":null},{"url":null,"slug":"dop-off-policy-multi-agent-decomposed-policy","title":"DOP: Off-Policy Multi-Agent Decomposed Policy Gradients","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fsv-learning-to-factorize-soft-value-function","title":"FSV: Learning to Factorize Soft Value Function for Cooperative Multi-Agent Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"rmix-risk-sensitive-multi-agent-reinforcement","title":"RMIX: Risk-Sensitive Multi-Agent Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"spatially-structured-recurrent-modules","title":"Spatially Structured Recurrent Modules","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"scc-an-efficient-deep-reinforcement-learning","title":"SCC: an efficient deep reinforcement learning agent mastering the game of StarCraft II","date":"2020-12-24","arxiv_id":"2012.13169","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-the-beginning-of","title":"Reinforcement Learning for the Beginning of Starcraft II Game","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"uneven-universal-value-exploration-for-multi-1","title":"UneVEn: Universal Value Exploration for Multi-Agent Reinforcement Learning","date":"2020-10-06","arxiv_id":"2010.02974","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatially-structured-recurrent-modules-1","title":"Spatially Structured Recurrent Modules","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-linear-value-1","title":"Towards Understanding Linear Value Decomposition in Cooperative Multi-Agent Q-Learning","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-with-graph","title":"BGC: Multi-Agent Group Belief with Graph Clustering","date":"2020-08-20","arxiv_id":"2008.08808","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchial-reinforcement-learning-in","title":"Hierarchical Reinforcement Learning in StarCraft II with Human Expertise in Subgoals Selection","date":"2020-08-08","arxiv_id":"2008.03444","repositories_listed":0,"syntology":null},{"url":null,"slug":"s2rms-spatially-structured-recurrent-modules","title":"S2RMs: Spatially Structured Recurrent Modules","date":"2020-07-13","arxiv_id":"2007.06533","repositories_listed":0,"syntology":null},{"url":null,"slug":"starcraft-ii-build-order-optimization-using","title":"StarCraft II Build Order Optimization using Deep Reinforcement Learning and Monte-Carlo Tree Search","date":"2020-06-12","arxiv_id":"2006.10525","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-pragmatic-reasoning","title":"Incorporating Pragmatic Reasoning Communication into Emergent Language","date":"2020-06-07","arxiv_id":"2006.04109","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-linear-value","title":"Towards Understanding Cooperative Multi-Agent Q-Learning with Value Factorization","date":"2020-05-31","arxiv_id":"2006.00587","repositories_listed":0,"syntology":null},{"url":null,"slug":"f2a2-flexible-fully-decentralized-approximate","title":"F2A2: Flexible Fully-decentralized Approximate Actor-critic for Cooperative Multi-agent Reinforcement Learning","date":"2020-04-17","arxiv_id":"2004.11145","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-theorem-for-latent-games-or-how-i","title":"A Limited-Capacity Minimax Theorem for Non-Convex Games or: How I Learned to Stop Worrying about Mixed-Nash and Love Neural Nets","date":"2020-02-14","arxiv_id":"2002.05820","repositories_listed":0,"syntology":null},{"url":null,"slug":"heterogeneous-learning-from-demonstration","title":"Heterogeneous Learning from Demonstration","date":"2020-01-27","arxiv_id":"2001.09569","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-narration-based-reward-shaping-approach","title":"A Narration-based Reward Shaping Approach using Grounded Natural Language Commands","date":"2019-10-31","arxiv_id":"1911.00497","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-a-behavioral-repertoire-from","title":"Learning a Behavioral Repertoire from Demonstrations","date":"2019-07-05","arxiv_id":"1907.03046","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-relational","title":"Deep reinforcement learning with relational inductive biases","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"190602671","title":"Grounding Natural Language Commands to StarCraft II Game States for Narration-Guided Reinforcement Learning","date":"2019-04-24","arxiv_id":"1906.02671","repositories_listed":0,"syntology":null},{"url":null,"slug":"alphastar-an-evolutionary-computation","title":"AlphaStar: An Evolutionary Computation Perspective","date":"2019-02-05","arxiv_id":"1902.01724","repositories_listed":0,"syntology":null},{"url":null,"slug":"dungeon-crawl-stone-soup-as-an-evaluation","title":"Dungeon Crawl Stone Soup as an Evaluation Domain for Artificial Intelligence","date":"2019-02-05","arxiv_id":"1902.01769","repositories_listed":0,"syntology":null},{"url":null,"slug":"modular-architecture-for-starcraft-ii-with","title":"Modular Architecture for StarCraft II with Deep Reinforcement Learning","date":"2018-11-08","arxiv_id":"1811.03555","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-reinforcement-learning-for-full-length","title":"On Reinforcement Learning for Full-length Game of StarCraft","date":"2018-09-23","arxiv_id":"1809.09095","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-advantage-actor-critic-agent-for","title":"Asynchronous Advantage Actor-Critic Agent for Starcraft II","date":"2018-07-22","arxiv_id":"1807.08217","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-deep-reinforcement-learning","title":"Towards a Deep Reinforcement Learning Approach for Tower Line Wars","date":"2017-12-17","arxiv_id":"1712.06180","repositories_listed":0,"syntology":null}],"record_sha256":"c60566b824e31d493fdcce357f259bd578cec09650ffb46b366502241e51ae7e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}