{"url":"/method/local-sgd","slug":"local-sgd","name":"Local SGD","full_name":"Local SGD","full_name_withheld":false,"description_markdown":"**Local SGD** is a distributed training technique that runs [SGD](https://paperswithcode.com/method/sgd) independently in parallel on different workers and averages the sequences only once in a while.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Local SGD Converges Fast and Communicates Little","paper":"/paper/local-sgd-converges-fast-and-communicates","first_author":"Sebastian U. Stich","n_authors":1,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/local-sgd-converges-fast-and-communicates"},"source":{"url":"https://arxiv.org/abs/1805.09767v3","title":"Local SGD Converges Fast and Communicates Little","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Data Parallel Methods","url":"/methods/category/data-parallel-methods","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Distributed Methods","url":"/methods/category/distributed-methods","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Optimization","url":"/methods/category/optimization","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Stochastic Optimization","url":"/methods/category/stochastic-optimization","pwc_aliases":[]}],"n_papers_tagged":69,"archive_num_papers":69,"papers_newest_first":[{"paper":null,"title":"DES-LOC: Desynced Low Communication Adaptive Optimizers for Training Foundation Models","date":"2025-05-28","arxiv_id":"2505.22549","n_code_links":0,"syntology":null},{"paper":null,"title":"Sharp Gaussian approximations for Decentralized Federated Learning","date":"2025-05-12","arxiv_id":"2505.08125","n_code_links":0,"syntology":null},{"paper":null,"title":"Streaming Federated Learning with Markovian Data","date":"2025-03-24","arxiv_id":"2503.18807","n_code_links":0,"syntology":null},{"paper":"/paper/edit-a-local-sgd-based-efficient-distributed","title":"EDiT: A Local-SGD-Based Efficient Distributed Training Method for Large Language Models","date":"2024-12-10","arxiv_id":"2412.07210","n_code_links":2,"syntology":null},{"paper":null,"title":"Collaborative and Efficient Personalization with Mixtures of Adaptors","date":"2024-10-04","arxiv_id":"2410.03497","n_code_links":0,"syntology":null},{"paper":null,"title":"Does Worst-Performing Agent Lead the Pack? Analyzing Agent Dynamics in Unified Distributed SGD","date":"2024-09-26","arxiv_id":"2409.17499","n_code_links":0,"syntology":null},{"paper":null,"title":"Convergence of Distributed Adaptive Optimization with Local Updates","date":"2024-09-20","arxiv_id":"2409.13155","n_code_links":0,"syntology":null},{"paper":null,"title":"Exploring Scaling Laws for Local SGD in Large Language Model Training","date":"2024-09-20","arxiv_id":"2409.13198","n_code_links":0,"syntology":null},{"paper":null,"title":"Communication-Efficient Adaptive Batch Size Strategies for Distributed Local Gradient Methods","date":"2024-06-20","arxiv_id":"2406.13936","n_code_links":0,"syntology":null},{"paper":null,"title":"Local Methods with Adaptivity via Scaling","date":"2024-06-02","arxiv_id":"2406.00846","n_code_links":0,"syntology":null},{"paper":null,"title":"The Limits and Potentials of Local SGD for Distributed Heterogeneous Learning with Intermittent Communication","date":"2024-05-19","arxiv_id":"2405.11667","n_code_links":0,"syntology":null},{"paper":"/paper/communication-efficient-and-provable","title":"Communication Efficient and Provable Federated Unlearning","date":"2024-01-19","arxiv_id":"2401.11018","n_code_links":1,"syntology":null},{"paper":"/paper/can-we-learn-communication-efficient","title":"Can We Learn Communication-Efficient Optimizers?","date":"2023-12-02","arxiv_id":"2312.02204","n_code_links":1,"syntology":null},{"paper":null,"title":"Asynchronous SGD on Graphs: a Unified Framework for Asynchronous Decentralized and Federated Optimization","date":"2023-11-01","arxiv_id":"2311.00465","n_code_links":0,"syntology":null},{"paper":"/paper/a-quadratic-synchronization-rule-for","title":"A Quadratic Synchronization Rule for Distributed Deep Learning","date":"2023-10-22","arxiv_id":"2310.14423","n_code_links":1,"syntology":{"ran":5,"of":6,"unverified":1,"pointer_only":0}},{"paper":null,"title":"Asynchronous Federated Learning with Incentive Mechanism Based on Contract Theory","date":"2023-10-10","arxiv_id":"2310.06448","n_code_links":0,"syntology":null},{"paper":null,"title":"Stability and Generalization for Minibatch SGD and Local SGD","date":"2023-10-02","arxiv_id":"2310.01139","n_code_links":0,"syntology":null},{"paper":null,"title":"Global Convergence Analysis of Local SGD for Two-layer Neural Network without Overparameterization","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Preconditioned Federated Learning","date":"2023-09-20","arxiv_id":"2309.11378","n_code_links":0,"syntology":null},{"paper":null,"title":"FedYolo: Augmenting Federated Learning with Pretrained Transformers","date":"2023-07-10","arxiv_id":"2307.04905","n_code_links":0,"syntology":null},{"paper":null,"title":"Semi-Asynchronous Federated Edge Learning Mechanism via Over-the-air Computation","date":"2023-05-06","arxiv_id":"2305.04066","n_code_links":0,"syntology":null},{"paper":"/paper/why-and-when-does-local-sgd-generalize-better","title":"Why (and When) does Local SGD Generalize Better than SGD?","date":"2023-03-02","arxiv_id":"2303.01215","n_code_links":1,"syntology":{"ran":3,"of":8,"unverified":5,"pointer_only":0}},{"paper":null,"title":"$z$-SignFedAvg: A Unified Stochastic Sign-based Compression for Federated Learning","date":"2023-02-06","arxiv_id":"2302.02589","n_code_links":0,"syntology":null},{"paper":"/paper/when-do-curricula-work-in-federated-learning","title":"When Do Curricula Work in Federated Learning?","date":"2022-12-24","arxiv_id":"2212.12712","n_code_links":1,"syntology":null},{"paper":null,"title":"STSyn: Speeding Up Local SGD with Straggler-Tolerant Synchronization","date":"2022-10-06","arxiv_id":"2210.03521","n_code_links":0,"syntology":null},{"paper":null,"title":"On the Stability Analysis of Open Federated Learning Systems","date":"2022-09-25","arxiv_id":"2209.12307","n_code_links":0,"syntology":null},{"paper":"/paper/fedaug-reducing-the-local-learning-bias","title":"FedBR: Improving Federated Learning on Heterogeneous Data via Local Learning Bias Reduction","date":"2022-05-26","arxiv_id":"2205.13462","n_code_links":1,"syntology":{"ran":9,"of":15,"unverified":6,"pointer_only":0}},{"paper":null,"title":"Federated Random Reshuffling with Compression and Variance Reduction","date":"2022-05-08","arxiv_id":"2205.03914","n_code_links":0,"syntology":null},{"paper":null,"title":"Federated Stochastic Primal-dual Learning with Differential Privacy","date":"2022-04-26","arxiv_id":"2204.12284","n_code_links":0,"syntology":null},{"paper":null,"title":"ImageNet Challenging Classification with the Raspberry Pi: An Incremental Local Stochastic Gradient Descent Algorithm","date":"2022-03-21","arxiv_id":"2203.11853","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/federated-learning","name":"Federated Learning","papers":27},{"task":"/task/distributed-optimization","name":"Distributed Optimization","papers":11},{"task":"/task/image-classification","name":"Image Classification","papers":4},{"task":"/task/language-modeling","name":"Language Modeling","papers":4},{"task":"/task/language-modelling","name":"Language Modelling","papers":4},{"task":"/task/image-classification","name":"image-classification","papers":4},{"task":"/task/edge-computing","name":"Edge-computing","papers":3},{"task":"/task/stochastic-optimization","name":"Stochastic Optimization","papers":3},{"task":"/task/machine-learning","name":"BIG-bench Machine Learning","papers":2},{"task":"/task/blocking","name":"Blocking","papers":2},{"task":"/task/privacy-preserving","name":"Privacy Preserving","papers":2},{"task":"/task/2k","name":"2k","papers":1},{"task":null,"name":"CPU","papers":1},{"task":"/task/deep-learning","name":"Deep Learning","papers":1},{"task":"/task/domain-generalization","name":"Domain Generalization","papers":1},{"task":null,"name":"GPU","papers":1},{"task":"/task/large-language-model","name":"Large Language Model","papers":1},{"task":"/task/machine-translation","name":"Machine Translation","papers":1},{"task":"/task/multi-task-learning","name":"Multi-Task Learning","papers":1},{"task":"/task/personalized-federated-learning","name":"Personalized Federated Learning","papers":1}],"tasks_shown":20,"n_tasks":23,"usage_by_year":[{"year":"2018","papers":1},{"year":"2019","papers":5},{"year":"2020","papers":11},{"year":"2021","papers":19},{"year":"2022","papers":10},{"year":"2023","papers":11},{"year":"2024","papers":9},{"year":"2025","papers":3}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/local-sgd"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}