{"url":"/method/gshard","slug":"gshard","name":"GShard","full_name":"GShard","full_name_withheld":false,"description_markdown":"**GShard** is a intra-layer parallel distributed method. It consists of set of simple APIs for annotations, and a compiler extension in XLA for automatic parallelization.","description_state":"present","introduced_year":null,"introduced_by":{"title":"GShard: Scaling Giant Models with Conditional Computation and Automatic Sharding","paper":"/paper/gshard-scaling-giant-models-with-conditional","first_author":"Dmitry Lepikhin","n_authors":9,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/gshard-scaling-giant-models-with-conditional"},"source":{"url":"https://arxiv.org/abs/2006.16668v1","title":"GShard: Scaling Giant Models with Conditional Computation and Automatic Sharding","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Intra-Layer Parallel","url":"/methods/category/intra-layer-parallel","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Model Parallel Methods","url":"/methods/category/model-parallel-methods","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Distributed Methods","url":"/methods/category/distributed-methods","pwc_aliases":[]}],"n_papers_tagged":6,"archive_num_papers":6,"papers_newest_first":[{"paper":"/paper/deepseekmoe-towards-ultimate-expert","title":"DeepSeekMoE: Towards Ultimate Expert Specialization in Mixture-of-Experts Language Models","date":"2024-01-11","arxiv_id":"2401.06066","n_code_links":2,"syntology":{"ran":6,"of":7,"unverified":1,"pointer_only":0}},{"paper":null,"title":"Mixture-of-Experts with Expert Choice Routing","date":"2022-02-18","arxiv_id":"2202.09368","n_code_links":0,"syntology":null},{"paper":null,"title":"Scaling End-to-End Models for Large-Scale Multilingual ASR","date":"2021-04-30","arxiv_id":"2104.14830","n_code_links":0,"syntology":null},{"paper":null,"title":"Carbon Emissions and Large Neural Network Training","date":"2021-04-21","arxiv_id":"2104.10350","n_code_links":0,"syntology":null},{"paper":null,"title":"Compression of Deep Learning Models for Text: A Survey","date":"2020-08-12","arxiv_id":"2008.05221","n_code_links":0,"syntology":null},{"paper":"/paper/gshard-scaling-giant-models-with-conditional","title":"GShard: Scaling Giant Models with Conditional Computation and Automatic Sharding","date":"2020-06-30","arxiv_id":"2006.16668","n_code_links":2,"syntology":{"ran":9,"of":10,"unverified":1,"pointer_only":1}}],"papers_shown":6,"tasks":[{"task":"/task/mixture-of-experts","name":"Mixture-of-Experts","papers":3},{"task":"/task/deep-learning","name":"Deep Learning","papers":1},{"task":"/task/information-retrieval","name":"Information Retrieval","papers":1},{"task":"/task/knowledge-distillation","name":"Knowledge Distillation","papers":1},{"task":"/task/language-modelling","name":"Language Modelling","papers":1},{"task":"/task/large-language-model","name":"Large Language Model","papers":1},{"task":"/task/machine-translation","name":"Machine Translation","papers":1},{"task":"/task/multi-task-learning","name":"Multi-Task Learning","papers":1},{"task":"/task/architecture-search","name":"Neural Architecture Search","papers":1},{"task":"/task/2048","name":"Playing the Game of 2048","papers":1},{"task":"/task/quantization","name":"Quantization","papers":1},{"task":"/task/retrieval","name":"Retrieval","papers":1},{"task":"/task/scheduling","name":"Scheduling","papers":1},{"task":"/task/survey","name":"Survey","papers":1},{"task":"/task/tensor-decomposition","name":"Tensor Decomposition","papers":1},{"task":"/task/translation","name":"Translation","papers":1}],"tasks_shown":16,"n_tasks":16,"usage_by_year":[{"year":"2020","papers":2},{"year":"2021","papers":2},{"year":"2022","papers":1},{"year":"2024","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/gshard"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}