{"url":"/method/powersgd","slug":"powersgd","name":"PowerSGD","full_name":"PowerSGD","full_name_withheld":false,"description_markdown":"**PowerSGD** is a distributed optimization technique that computes a low-rank approximation of the gradient using a generalized power iteration (known as subspace iteration). The approximation is computationally light-weight, avoiding any prohibitively expensive Singular Value Decomposition. To improve the quality of the efficient approximation, the authors warm-start the power iteration by reusing the approximation from the previous optimization step.","description_state":"present","introduced_year":null,"introduced_by":{"title":"PowerSGD: Practical Low-Rank Gradient Compression for Distributed Optimization","paper":"/paper/powersgd-practical-low-rank-gradient","first_author":"Thijs Vogels","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/powersgd-practical-low-rank-gradient"},"source":{"url":"https://arxiv.org/abs/1905.13727v3","title":"PowerSGD: Practical Low-Rank Gradient Compression for Distributed Optimization","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Data Parallel Methods","url":"/methods/category/data-parallel-methods","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Distributed Methods","url":"/methods/category/distributed-methods","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Optimization","url":"/methods/category/optimization","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Stochastic Optimization","url":"/methods/category/stochastic-optimization","pwc_aliases":[]}],"n_papers_tagged":3,"archive_num_papers":3,"papers_newest_first":[{"paper":"/paper/practical-low-rank-communication-compression","title":"Practical Low-Rank Communication Compression in Decentralized Deep Learning","date":"2020-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/powergossip-practical-low-rank-communication","title":"PowerGossip: Practical Low-Rank Communication Compression in Decentralized Deep Learning","date":"2020-08-04","arxiv_id":"2008.01425","n_code_links":1,"syntology":{"ran":3,"of":3,"unverified":0,"pointer_only":2}},{"paper":"/paper/powersgd-practical-low-rank-gradient","title":"PowerSGD: Practical Low-Rank Gradient Compression for Distributed Optimization","date":"2019-05-31","arxiv_id":"1905.13727","n_code_links":1,"syntology":{"ran":3,"of":3,"unverified":0,"pointer_only":0}}],"papers_shown":3,"tasks":[{"task":"/task/deep-learning","name":"Deep Learning","papers":2},{"task":"/task/distributed-optimization","name":"Distributed Optimization","papers":1}],"tasks_shown":2,"n_tasks":2,"usage_by_year":[{"year":"2019","papers":1},{"year":"2020","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/powersgd"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}