{"url":"/method/zero","slug":"zero","name":"ZeRO","full_name":"ZeRO","full_name_withheld":false,"description_markdown":"**Zero Redundancy Optimizer (ZeRO)** is a sharded data parallel method for distributed training. ZeRODP removes the memory state redundancies across data-parallel processes by partitioning the model states instead of replicating them, and it retains the compute/communication efficiency by retaining the computational granularity and communication volume of DP using a dynamic communication schedule during training.","description_state":"present","introduced_year":null,"introduced_by":{"title":"ZeRO: Memory Optimizations Toward Training Trillion Parameter Models","paper":"/paper/zero-memory-optimization-towards-training-a","first_author":"Samyam Rajbhandari","n_authors":4,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/zero-memory-optimization-towards-training-a"},"source":{"url":"https://arxiv.org/abs/1910.02054v3","title":"ZeRO: Memory Optimizations Toward Training Trillion Parameter Models","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Sharded Data Parallel Methods","url":"/methods/category/sharded-data-parallel-methods","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Data Parallel Methods","url":"/methods/category/data-parallel-methods","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Distributed Methods","url":"/methods/category/distributed-methods","pwc_aliases":[]}],"n_papers_tagged":9,"archive_num_papers":9,"papers_newest_first":[{"paper":null,"title":"Memory Analysis on the Training Course of DeepSeek Models","date":"2025-02-11","arxiv_id":"2502.07846","n_code_links":0,"syntology":null},{"paper":null,"title":"Accelerating Large Language Model Training with Hybrid GPU-based Compression","date":"2024-09-04","arxiv_id":"2409.02423","n_code_links":0,"syntology":null},{"paper":null,"title":"A Study of Optimizations for Fine-tuning Large Language Models","date":"2024-06-04","arxiv_id":"2406.02290","n_code_links":0,"syntology":null},{"paper":null,"title":"Zero redundancy distributed learning with differential privacy","date":"2023-11-20","arxiv_id":"2311.11822","n_code_links":0,"syntology":null},{"paper":null,"title":"Dissecting the Runtime Performance of the Training, Fine-tuning, and Inference of Large Language Models","date":"2023-11-07","arxiv_id":"2311.03687","n_code_links":0,"syntology":null},{"paper":"/paper/remax-a-simple-effective-and-efficient-method","title":"ReMax: A Simple, Effective, and Efficient Reinforcement Learning Method for Aligning Large Language Models","date":"2023-10-16","arxiv_id":"2310.10505","n_code_links":3,"syntology":{"ran":8,"of":13,"unverified":5,"pointer_only":13}},{"paper":null,"title":"Rethinking Memory and Communication Cost for Efficient Large Language Model Training","date":"2023-10-09","arxiv_id":"2310.06003","n_code_links":0,"syntology":null},{"paper":"/paper/zero-extremely-efficient-collective","title":"ZeRO++: Extremely Efficient Collective Communication for Giant Model Training","date":"2023-06-16","arxiv_id":"2306.10209","n_code_links":1,"syntology":null},{"paper":"/paper/zero-memory-optimization-towards-training-a","title":"ZeRO: Memory Optimizations Toward Training Trillion Parameter Models","date":"2019-10-04","arxiv_id":"1910.02054","n_code_links":10,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}}],"papers_shown":9,"tasks":[{"task":null,"name":"GPU","papers":7},{"task":"/task/language-modelling","name":"Language Modelling","papers":3},{"task":"/task/language-modeling","name":"Language Modeling","papers":2},{"task":"/task/large-language-model","name":"Large Language Model","papers":2},{"task":"/task/quantization","name":"Quantization","papers":2},{"task":"/task/cross-lingual-document-classification","name":"Cross-Lingual Document Classification","papers":1},{"task":"/task/general-reinforcement-learning","name":"General Reinforcement Learning","papers":1},{"task":"/task/image-generation","name":"Image Generation","papers":1},{"task":"/task/mixture-of-experts","name":"Mixture-of-Experts","papers":1},{"task":"/task/privacy-preserving","name":"Privacy Preserving","papers":1},{"task":"/task/reinforcement-learning-2","name":"reinforcement-learning","papers":1}],"tasks_shown":11,"n_tasks":11,"usage_by_year":[{"year":"2019","papers":1},{"year":"2023","papers":5},{"year":"2024","papers":2},{"year":"2025","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/zero"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}