{"url":"/method/douzero","slug":"douzero","name":"DouZero","full_name":"DouZero","full_name_withheld":false,"description_markdown":"**DouZero** is an AI system for the card game DouDizhu that enhances traditional Monte-Carlo methods with deep neural networks, action encoding, and parallel actors. The [Q-network](https://paperswithcode.com/method/dqn) of DouZero consists of an [LSTM](https://paperswithcode.com/method/lstm) to encode historical actions and six layers of [MLP](https://paperswithcode.com/method/feedforward-network) with hidden dimension of 512. The network predicts a value for a given state-action pair based on the concatenated representation of action and state.","description_state":"present","introduced_year":null,"introduced_by":{"title":"DouZero: Mastering DouDizhu with Self-Play Deep Reinforcement Learning","paper":"/paper/douzero-mastering-doudizhu-with-self-play","first_author":"Daochen Zha","n_authors":7,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/douzero-mastering-doudizhu-with-self-play"},"source":{"url":"https://arxiv.org/abs/2106.06135v1","title":"DouZero: Mastering DouDizhu with Self-Play Deep Reinforcement Learning","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Reinforcement Learning","area_id":"reinforcement-learning","collection":"Card Game Models","url":"/methods/category/card-game-models","pwc_aliases":[]}],"n_papers_tagged":4,"archive_num_papers":4,"papers_newest_first":[{"paper":"/paper/alphadou-high-performance-end-to-end-doudizhu","title":"AlphaDou: High-Performance End-to-End Doudizhu AI Integrating Bidding","date":"2024-07-14","arxiv_id":"2407.10279","n_code_links":1,"syntology":null},{"paper":null,"title":"DouRN: Improving DouZero by Residual Neural Networks","date":"2024-03-21","arxiv_id":"2403.14102","n_code_links":0,"syntology":null},{"paper":"/paper/douzero-improving-doudizhu-ai-by-opponent","title":"DouZero+: Improving DouDizhu AI by Opponent Modeling and Coach-guided Learning","date":"2022-04-06","arxiv_id":"2204.02558","n_code_links":1,"syntology":null},{"paper":"/paper/douzero-mastering-doudizhu-with-self-play","title":"DouZero: Mastering DouDizhu with Self-Play Deep Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06135","n_code_links":1,"syntology":{"ran":3,"of":6,"unverified":3,"pointer_only":1}}],"papers_shown":4,"tasks":[{"task":"/task/deep-reinforcement-learning","name":"Deep Reinforcement Learning","papers":3},{"task":"/task/card-games","name":"Card Games","papers":2},{"task":"/task/game-of-poker","name":"Game of Poker","papers":1},{"task":"/task/multi-agent-reinforcement-learning","name":"Multi-agent Reinforcement Learning","papers":1},{"task":"/task/reinforcement-learning","name":"Reinforcement Learning","papers":1},{"task":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)","papers":1},{"task":"/task/reinforcement-learning-2","name":"reinforcement-learning","papers":1}],"tasks_shown":7,"n_tasks":7,"usage_by_year":[{"year":"2021","papers":1},{"year":"2022","papers":1},{"year":"2024","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/douzero"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}