{"url":"/method/zero-offload","slug":"zero-offload","name":"ZeRO-Offload","full_name":"ZeRO-Offload","full_name_withheld":false,"description_markdown":"ZeRO-Offload is a sharded data parallel method for distributed training. It exploits both CPU memory and compute for offloading, while offering a clear path towards efficiently scaling on multiple GPUs by working with [ZeRO-powered data parallelism](https://www.paperswithcode.com/method/zero). The symbiosis allows ZeRO-Offload to maintain a single copy of the optimizer states on the CPU memory regardless of the data parallel degree. Furthermore, it keeps the aggregate communication volume between GPU and CPU, as well as the aggregate CPU computation a constant regardless of data parallelism, allowing ZeRO-Offload to effectively utilize the linear increase in CPU compute with the increase in the data parallelism degree.","description_state":"present","introduced_year":null,"introduced_by":{"title":"ZeRO-Offload: Democratizing Billion-Scale Model Training","paper":"/paper/zero-offload-democratizing-billion-scale","first_author":"Jie Ren","n_authors":8,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/zero-offload-democratizing-billion-scale"},"source":{"url":"https://arxiv.org/abs/2101.06840v1","title":"ZeRO-Offload: Democratizing Billion-Scale Model Training","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Sharded Data Parallel Methods","url":"/methods/category/sharded-data-parallel-methods","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Data Parallel Methods","url":"/methods/category/data-parallel-methods","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Distributed Methods","url":"/methods/category/distributed-methods","pwc_aliases":[]}],"n_papers_tagged":4,"archive_num_papers":4,"papers_newest_first":[{"paper":null,"title":"Cost-Efficient LLM Training with Lifetime-Aware Tensor Offloading via GPUDirect Storage","date":"2025-06-06","arxiv_id":"2506.06472","n_code_links":0,"syntology":null},{"paper":null,"title":"ZenFlow: Enabling Stall-Free Offloading Training via Asynchronous Updates","date":"2025-05-18","arxiv_id":"2505.12242","n_code_links":0,"syntology":null},{"paper":null,"title":"ProTrain: Efficient LLM Training via Memory-Aware Techniques","date":"2024-06-12","arxiv_id":"2406.08334","n_code_links":0,"syntology":null},{"paper":"/paper/zero-offload-democratizing-billion-scale","title":"ZeRO-Offload: Democratizing Billion-Scale Model Training","date":"2021-01-18","arxiv_id":"2101.06840","n_code_links":3,"syntology":null}],"papers_shown":4,"tasks":[{"task":null,"name":"CPU","papers":4},{"task":null,"name":"GPU","papers":4},{"task":"/task/computational-efficiency","name":"Computational Efficiency","papers":1},{"task":"/task/large-language-model","name":"Large Language Model","papers":1},{"task":"/task/management","name":"Management","papers":1},{"task":"/task/model","name":"model","papers":1}],"tasks_shown":6,"n_tasks":6,"usage_by_year":[{"year":"2021","papers":1},{"year":"2024","papers":1},{"year":"2025","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/zero-offload"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}