{"url":"/method/zero-infinity","slug":"zero-infinity","name":"ZeRO-Infinity","full_name":"ZeRO-Infinity","full_name_withheld":false,"description_markdown":"**ZeRO-Infinity** is a sharded data parallel system that extends [ZeRO](https://paperswithcode.com/method/zero) with new innovations in heterogeneous memory access called the infinity offload engine. This allows ZeRO-Infinity to support massive model sizes on limited GPU resources by exploiting CPU and NVMe memory simultaneously. In addition, ZeRO-Infinity also introduces a novel GPU memory optimization technique called memory-centric tiling to support extremely large individual layers that would otherwise not fit in GPU memory even one layer at a time.","description_state":"present","introduced_year":null,"introduced_by":{"title":"ZeRO-Infinity: Breaking the GPU Memory Wall for Extreme Scale Deep Learning","paper":"/paper/zero-infinity-breaking-the-gpu-memory-wall","first_author":"Samyam Rajbhandari","n_authors":5,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/zero-infinity-breaking-the-gpu-memory-wall"},"source":{"url":"https://arxiv.org/abs/2104.07857v1","title":"ZeRO-Infinity: Breaking the GPU Memory Wall for Extreme Scale Deep Learning","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Sharded Data Parallel Methods","url":"/methods/category/sharded-data-parallel-methods","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Data Parallel Methods","url":"/methods/category/data-parallel-methods","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Distributed Methods","url":"/methods/category/distributed-methods","pwc_aliases":[]}],"n_papers_tagged":2,"archive_num_papers":2,"papers_newest_first":[{"paper":null,"title":"Cost-Efficient LLM Training with Lifetime-Aware Tensor Offloading via GPUDirect Storage","date":"2025-06-06","arxiv_id":"2506.06472","n_code_links":0,"syntology":null},{"paper":"/paper/zero-infinity-breaking-the-gpu-memory-wall","title":"ZeRO-Infinity: Breaking the GPU Memory Wall for Extreme Scale Deep Learning","date":"2021-04-16","arxiv_id":"2104.07857","n_code_links":0,"syntology":null}],"papers_shown":2,"tasks":[{"task":null,"name":"CPU","papers":2},{"task":null,"name":"GPU","papers":2},{"task":"/task/large-language-model","name":"Large Language Model","papers":1}],"tasks_shown":3,"n_tasks":3,"usage_by_year":[{"year":"2021","papers":1},{"year":"2025","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/zero-infinity"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}