{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/10","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":10,"pages_in_order":21,"rows_per_page":100,"rows":[901,1000],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/9","next":"/method/sgd/papers/11","papers":[{"paper":null,"slug":"dynamic-differential-privacy-preserving-sgd-1","title":"Dynamic Differential-Privacy Preserving SGD","date":"2021-10-30","arxiv_id":"2111.00173","n_code_links":0,"syntology":null},{"paper":null,"slug":"does-momentum-help-a-sample-complexity","title":"Does Momentum Help? A Sample Complexity Analysis","date":"2021-10-29","arxiv_id":"2110.15547","n_code_links":0,"syntology":null},{"paper":"/paper/eigencurve-optimal-learning-rate-schedule-for-1","slug":"eigencurve-optimal-learning-rate-schedule-for-1","title":"Eigencurve: Optimal Learning Rate Schedule for SGD on Quadratic Objectives with Skewed Hessian Spectrums","date":"2021-10-27","arxiv_id":"2110.14109","n_code_links":1,"syntology":null},{"paper":null,"slug":"multilayer-lookahead-a-nested-version-of","title":"Multilayer Lookahead: a Nested Version of Lookahead","date":"2021-10-27","arxiv_id":"2110.14254","n_code_links":0,"syntology":null},{"paper":"/paper/exponential-graph-is-provably-efficient-for","slug":"exponential-graph-is-provably-efficient-for","title":"Exponential Graph is Provably Efficient for Decentralized Deep Training","date":"2021-10-26","arxiv_id":"2110.13363","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Bluefog-Lib/bluefog","bluefog-lib/neurips2021-exponential-graph"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"asynchronous-decentralized-distributed","title":"Asynchronous Decentralized Distributed Training of Acoustic Models","date":"2021-10-21","arxiv_id":"2110.11199","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-noise-adaptive-problem-adaptive","title":"Towards Noise-adaptive, Problem-adaptive (Accelerated) Stochastic Gradient Descent","date":"2021-10-21","arxiv_id":"2110.11442","n_code_links":0,"syntology":null},{"paper":null,"slug":"minibatch-vs-local-sgd-with-shuffling-tight-1","title":"Minibatch vs Local SGD with Shuffling: Tight Convergence Bounds and Beyond","date":"2021-10-20","arxiv_id":"2110.10342","n_code_links":0,"syntology":null},{"paper":"/paper/training-deep-neural-networks-with-adaptive","slug":"training-deep-neural-networks-with-adaptive","title":"Training Deep Neural Networks with Adaptive Momentum Inspired by the Quadratic Optimization","date":"2021-10-18","arxiv_id":"2110.09057","n_code_links":1,"syntology":{"ran":13,"of":17,"n_ran_checked":7,"n_instrument":6,"unverified":4,"pointer_only":5,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 3 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","official":{"repos":["kentaroy47/vision-transformers-cifar10"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-compositional-generalization-with","slug":"improving-compositional-generalization-with","title":"Improving Compositional Generalization with Self-Training for Data-to-Text Generation","date":"2021-10-16","arxiv_id":"2110.08467","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-better-plasticity-stability-trade-off","title":"Towards Better Plasticity-Stability Trade-off in Incremental Learning: A Simple Linear Connector","date":"2021-10-15","arxiv_id":"2110.07905","n_code_links":0,"syntology":null},{"paper":null,"slug":"trade-offs-of-local-sgd-at-scale-an-empirical","title":"Trade-offs of Local SGD at Scale: An Empirical Study","date":"2021-10-15","arxiv_id":"2110.08133","n_code_links":0,"syntology":null},{"paper":"/paper/adaptive-elastic-training-for-sparse-deep","slug":"adaptive-elastic-training-for-sparse-deep","title":"Adaptive Elastic Training for Sparse Deep Learning on Heterogeneous Multi-GPU Servers","date":"2021-10-13","arxiv_id":"2110.07029","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-double-descent-of-random-features","title":"On the Double Descent of Random Features Models Trained with SGD","date":"2021-10-13","arxiv_id":"2110.06910","n_code_links":0,"syntology":null},{"paper":"/paper/sgd-x-a-benchmark-for-robust-generalization","slug":"sgd-x-a-benchmark-for-robust-generalization","title":"SGD-X: A Benchmark for Robust Generalization in Schema-Guided Dialogue Systems","date":"2021-10-13","arxiv_id":"2110.06800","n_code_links":1,"syntology":null},{"paper":null,"slug":"what-happens-after-sgd-reaches-zero-loss-a-1","title":"What Happens after SGD Reaches Zero Loss? --A Mathematical Framework","date":"2021-10-13","arxiv_id":"2110.06914","n_code_links":0,"syntology":null},{"paper":null,"slug":"last-iterate-risk-bounds-of-sgd-with-decaying","title":"Last Iterate Risk Bounds of SGD with Decaying Stepsize for Overparameterized Linear Regression","date":"2021-10-12","arxiv_id":"2110.06198","n_code_links":0,"syntology":null},{"paper":"/paper/not-all-noise-is-accounted-equally-how","slug":"not-all-noise-is-accounted-equally-how","title":"Not all noise is accounted equally: How differentially private learning benefits from large sampling rates","date":"2021-10-12","arxiv_id":"2110.06255","n_code_links":1,"syntology":null},{"paper":"/paper/the-role-of-permutation-invariance-in-linear-1","slug":"the-role-of-permutation-invariance-in-linear-1","title":"The Role of Permutation Invariance in Linear Mode Connectivity of Neural Networks","date":"2021-10-12","arxiv_id":"2110.06296","n_code_links":1,"syntology":null},{"paper":"/paper/momentum-centering-and-asynchronous-update","slug":"momentum-centering-and-asynchronous-update","title":"Momentum Centering and Asynchronous Update for Adaptive Gradient Methods","date":"2021-10-11","arxiv_id":"2110.05454","n_code_links":2,"syntology":null},{"paper":null,"slug":"frequency-aware-sgd-for-efficient-embedding-1","title":"Frequency-aware SGD for Efficient Embedding Learning with Provable Benefits","date":"2021-10-10","arxiv_id":"2110.04844","n_code_links":0,"syntology":null},{"paper":null,"slug":"combining-differential-privacy-and-byzantine","title":"Combining Differential Privacy and Byzantine Resilience in Distributed SGD","date":"2021-10-08","arxiv_id":"2110.03991","n_code_links":0,"syntology":null},{"paper":null,"slug":"momentum-doesn-t-change-the-implicit-bias","title":"Does Momentum Change the Implicit Regularization on Separable Data?","date":"2021-10-08","arxiv_id":"2110.03891","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-generalization-of-models-trained-with","title":"On the Generalization of Models Trained with SGD: Information-Theoretic Bounds and Implications","date":"2021-10-07","arxiv_id":"2110.03128","n_code_links":0,"syntology":null},{"paper":null,"slug":"spectral-bias-in-practice-the-role-of","title":"Spectral Bias in Practice: The Role of Function Frequency in Generalization","date":"2021-10-06","arxiv_id":"2110.02424","n_code_links":0,"syntology":null},{"paper":"/paper/rapid-training-of-deep-neural-networks","slug":"rapid-training-of-deep-neural-networks","title":"Rapid training of deep neural networks without skip connections or normalization layers using Deep Kernel Shaping","date":"2021-10-05","arxiv_id":"2110.01765","n_code_links":2,"syntology":null},{"paper":null,"slug":"s2-reducer-high-performance-sparse","title":"S2 Reducer: High-Performance Sparse Communication to Accelerate Distributed Deep Learning","date":"2021-10-05","arxiv_id":"2110.02140","n_code_links":0,"syntology":null},{"paper":"/paper/effectiveness-of-optimization-algorithms-in","slug":"effectiveness-of-optimization-algorithms-in","title":"Effectiveness of Optimization Algorithms in Deep Image Classification","date":"2021-10-04","arxiv_id":"2110.01598","n_code_links":1,"syntology":null},{"paper":null,"slug":"global-convergence-and-stability-of","title":"Global Convergence and Stability of Stochastic Gradient Descent","date":"2021-10-04","arxiv_id":"2110.01663","n_code_links":0,"syntology":null},{"paper":null,"slug":"fed-lamb-layerwise-and-dimensionwise-locally","title":"Layer-wise and Dimension-wise Locally Adaptive Federated Learning","date":"2021-10-01","arxiv_id":"2110.00532","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-class-of-short-term-recurrence-anderson","title":"A Class of Short-term Recurrence Anderson Mixing Methods and Their Applications","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-general-analysis-of-example-selection-for","title":"A General Analysis of Example-Selection for Stochastic Gradient Descent","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-novel-convergence-analysis-for-the","title":"A Novel Convergence Analysis for the Stochastic Proximal Point Algorithm","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-practical-pac-bayes-generalisation-bound","title":"A Practical PAC-Bayes Generalisation Bound for Deep Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-inertia-disentangling-the-effects-of","title":"Adaptive Inertia: Disentangling the Effects of Adaptive Learning Rate and Momentum","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"boosting-the-confidence-of-near-tight","title":"Boosting the Confidence of Near-Tight Generalization Bounds for Uniformly Stable Randomized Algorithms","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"determining-the-ethno-nationality-of-writers","title":"Determining the Ethno-nationality of Writers Using Written English Text","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"directional-bias-helps-stochastic-gradient","title":"Directional Bias Helps Stochastic Gradient Descent to Generalize in Nonparametric Model","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-second-order-optimization-for-deep","title":"Efficient Second-Order Optimization for Deep Learning with Kernel Machines","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"how-to-improve-sample-complexity-of-sgd-over","title":"How to Improve Sample Complexity of SGD over Highly Dependent Data?","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-local-sgd-for-federated-learning-with","title":"Hybrid Local SGD for Federated Learning with Heterogeneous Communications","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"image-functions-in-neural-networks-a","title":"Image Functions In Neural Networks: A Perspective On Generalization","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/learning-pruning-friendly-networks-via-frank","slug":"learning-pruning-friendly-networks-via-frank","title":"Learning Pruning-Friendly Networks via Frank-Wolfe: One-Shot, Any-Sparsity, And No Retraining","date":"2021-09-29","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-rate-grafting-transferability-of","title":"Learning Rate Grafting: Transferability of Optimizer Tuning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"linear-backpropagation-leads-to-faster","title":"Linear Backpropagation Leads to Faster Convergence","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"logarithmic-landscape-and-power-law-escape-1","title":"Logarithmic landscape and power-law escape rate of SGD","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-nonconvex-continual","title":"On the Convergence of Nonconvex Continual Learning with Adaptive Learning Rate","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"optimization-and-adaptive-generalization-of","title":"Optimization and Adaptive Generalization of Three layer Neural Networks","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"orthogonalising-gradients-to-speedup-neural","title":"Orthogonalising gradients to speedup neural network optimisation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"rethinking-the-limiting-dynamics-of-sgd-1","title":"Rethinking the limiting dynamics of SGD: modified loss, phase space oscillations, and anomalous diffusion","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-can-converge-to-local-maxima","title":"SGD Can Converge to Local Maxima","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"slim-qn-a-stochastic-light-momentumized-quasi","title":"SLIM-QN: A Stochastic, Light, Momentumized Quasi-Newton Optimizer for Deep Neural Networks","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"spsc-a-fast-and-provable-algorithm-for","title":"SpSC: A Fast and Provable Algorithm for Sampling-Based GNN Training","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"squeezing-sgd-parallelization-performance-in","title":"Squeezing SGD Parallelization Performance in Distributed Training Using Delayed Averaging","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/stochastic-training-is-not-necessary-for","slug":"stochastic-training-is-not-necessary-for","title":"Stochastic Training is Not Necessary for Generalization","date":"2021-09-29","arxiv_id":"2109.14119","n_code_links":1,"syntology":null},{"paper":null,"slug":"two-regimes-of-generalization-for-non-linear","title":"Two Regimes of Generalization for Non-Linear Metric Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/iglu-efficient-gcn-training-via-lazy-updates","slug":"iglu-efficient-gcn-training-via-lazy-updates","title":"IGLU: Efficient GCN Training via Lazy Updates","date":"2021-09-28","arxiv_id":"2109.13995","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["sdeepaknarayanan/iglu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"accelerated-pdes-for-construction-and","title":"Accelerated PDEs for Construction and Theoretical Analysis of an SGD Extension","date":"2021-09-27","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/unrolling-sgd-understanding-factors","slug":"unrolling-sgd-understanding-factors","title":"Unrolling SGD: Understanding Factors Influencing Machine Unlearning","date":"2021-09-27","arxiv_id":"2109.13398","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":10,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cleverhans-lab/unrolling-sgd"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/curvature-injected-adaptive-momentum","slug":"curvature-injected-adaptive-momentum","title":"AdaInject: Injection Based Adaptive Gradient Descent Optimizers for Convolutional Neural Networks","date":"2021-09-26","arxiv_id":"2109.12504","n_code_links":1,"syntology":null},{"paper":null,"slug":"mixnn-protection-of-federated-learning","title":"MixNN: Protection of Federated Learning Against Inference Attacks by Mixing Neural Network Layers","date":"2021-09-26","arxiv_id":"2109.12550","n_code_links":0,"syntology":null},{"paper":"/paper/quantization-for-distributed-optimization","slug":"quantization-for-distributed-optimization","title":"Unbiased Single-scale and Multi-scale Quantizers for Distributed Optimization","date":"2021-09-26","arxiv_id":"2109.12497","n_code_links":1,"syntology":null},{"paper":null,"slug":"nanobatch-dpsgd-exploring-differentially","title":"NanoBatch Privacy: Enabling fast Differentially Private learning on the IPU","date":"2021-09-24","arxiv_id":"2109.12191","n_code_links":0,"syntology":null},{"paper":"/paper/neural-network-relief-a-pruning-algorithm","slug":"neural-network-relief-a-pruning-algorithm","title":"Neural network relief: a pruning algorithm based on neural activity","date":"2021-09-22","arxiv_id":"2109.10795","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-learn-to-explain-siamese-networks","title":"Self-learn to Explain Siamese Networks Robustly","date":"2021-09-15","arxiv_id":"2109.07371","n_code_links":0,"syntology":null},{"paper":null,"slug":"toward-communication-efficient-adaptive","title":"Toward Communication Efficient Adaptive Gradient Method","date":"2021-09-10","arxiv_id":"2109.05109","n_code_links":0,"syntology":null},{"paper":null,"slug":"sample-and-communication-efficient","title":"Sample and Communication-Efficient Decentralized Actor-Critic Algorithms with Finite-Time Analysis","date":"2021-09-08","arxiv_id":"2109.03699","n_code_links":0,"syntology":null},{"paper":"/paper/coco-denoiser-using-co-coercivity-for","slug":"coco-denoiser-using-co-coercivity-for","title":"COCO Denoiser: Using Co-Coercivity for Variance Reduction in Stochastic Convex Optimization","date":"2021-09-07","arxiv_id":"2109.03207","n_code_links":1,"syntology":null},{"paper":null,"slug":"revisiting-recursive-least-squares-for","title":"Revisiting Recursive Least Squares for Training Deep Neural Networks","date":"2021-09-07","arxiv_id":"2109.03220","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-subgradient-descent-on-a-generic","title":"Stochastic Subgradient Descent on a Generic Definable Function Converges to a Minimizer","date":"2021-09-06","arxiv_id":"2109.02455","n_code_links":0,"syntology":null},{"paper":null,"slug":"statistical-estimation-and-inference-via","title":"Statistical Estimation and Inference via Local SGD in Federated Learning","date":"2021-09-03","arxiv_id":"2109.01326","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-minimax-complexity-of-distributed","title":"The Minimax Complexity of Distributed Optimization","date":"2021-09-01","arxiv_id":"2109.00534","n_code_links":0,"syntology":null},{"paper":"/paper/using-a-one-dimensional-parabolic-model-of","slug":"using-a-one-dimensional-parabolic-model-of","title":"Using a one dimensional parabolic model of the full-batch loss to estimate learning rates during training","date":"2021-08-31","arxiv_id":"2108.13880","n_code_links":1,"syntology":null},{"paper":null,"slug":"byzantine-fault-tolerance-in-federated-local","title":"Byzantine Fault-Tolerance in Federated Local SGD under 2f-Redundancy","date":"2021-08-26","arxiv_id":"2108.11769","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-can-increased-randomness-in-stochastic","title":"Shift-Curvature, SGD, and Generalization","date":"2021-08-21","arxiv_id":"2108.09507","n_code_links":0,"syntology":null},{"paper":"/paper/communication-efficient-federated-learning-8","slug":"communication-efficient-federated-learning-8","title":"EDEN: Communication-Efficient and Robust Distributed Mean Estimation for Federated Learning","date":"2021-08-19","arxiv_id":"2108.08842","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["amitport/eden-distributed-mean-estimation"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"compressing-gradients-by-exploiting-temporal","title":"Compressing gradients by exploiting temporal correlation in momentum-SGD","date":"2021-08-17","arxiv_id":"2108.07827","n_code_links":0,"syntology":null},{"paper":null,"slug":"fedpage-a-fast-local-stochastic-gradient","title":"FedPAGE: A Fast Local Stochastic Gradient Method for Communication-Efficient Federated Learning","date":"2021-08-10","arxiv_id":"2108.04755","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-benefits-of-implicit-regularization-from","title":"The Benefits of Implicit Regularization from SGD in Least Squares Problems","date":"2021-08-10","arxiv_id":"2108.04552","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-hyperparameter-optimization-for-1","slug":"efficient-hyperparameter-optimization-for-1","title":"Efficient Hyperparameter Optimization for Differentially Private Deep Learning","date":"2021-08-09","arxiv_id":"2108.03888","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-hyperparameters-in-stochastic-gradient","title":"On the Hyperparameters in Stochastic Gradient Descent with Momentum","date":"2021-08-09","arxiv_id":"2108.03947","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-power-of-differentiable-learning","title":"On the Power of Differentiable Learning versus PAC and SQ Learning","date":"2021-08-09","arxiv_id":"2108.04190","n_code_links":0,"syntology":null},{"paper":null,"slug":"unified-regularity-measures-for-sample-wise","title":"Unified Regularity Measures for Sample-wise Learning and Generalization","date":"2021-08-09","arxiv_id":"2108.03913","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-subgradient-descent-escapes-active","title":"Stochastic Subgradient Descent Escapes Active Strict Saddles on Weakly Convex Functions","date":"2021-08-04","arxiv_id":"2108.02072","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-scale-differentially-private-bert","title":"Large-Scale Differentially Private BERT","date":"2021-08-03","arxiv_id":"2108.01624","n_code_links":0,"syntology":null},{"paper":null,"slug":"rethinking-gradient-sparsification-as-total","title":"Rethinking gradient sparsification as total error minimization","date":"2021-08-02","arxiv_id":"2108.00951","n_code_links":0,"syntology":null},{"paper":null,"slug":"provably-efficient-lottery-ticket-discovery","title":"How much pre-training is enough to discover a good subnetwork?","date":"2021-07-31","arxiv_id":"2108.00259","n_code_links":0,"syntology":null},{"paper":null,"slug":"dq-sgd-dynamic-quantization-in-sgd-for","title":"DQ-SGD: Dynamic Quantization in SGD for Communication-Efficient Distributed Learning","date":"2021-07-30","arxiv_id":"2107.14575","n_code_links":0,"syntology":null},{"paper":"/paper/manipulating-identical-filter-redundancy-for","slug":"manipulating-identical-filter-redundancy-for","title":"Manipulating Identical Filter Redundancy for Efficient Pruning on Deep and Complicated CNN","date":"2021-07-30","arxiv_id":"2107.14444","n_code_links":2,"syntology":null},{"paper":"/paper/decentralized-federated-learning-balancing","slug":"decentralized-federated-learning-balancing","title":"Decentralized Federated Learning: Balancing Communication and Computing Costs","date":"2021-07-26","arxiv_id":"2107.12048","n_code_links":1,"syntology":null},{"paper":null,"slug":"sgd-may-never-escape-saddle-points","title":"SGD with a Constant Large Learning Rate Can Converge to Local Maxima","date":"2021-07-25","arxiv_id":"2107.11774","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-general-sample-complexity-analysis-of","title":"A general sample complexity analysis of vanilla policy gradient","date":"2021-07-23","arxiv_id":"2107.11433","n_code_links":0,"syntology":null},{"paper":null,"slug":"local-sgd-optimizes-overparameterized-neural","title":"Local SGD Optimizes Overparameterized Neural Networks in Polynomial Time","date":"2021-07-22","arxiv_id":"2107.10868","n_code_links":0,"syntology":null},{"paper":null,"slug":"distribution-of-classification-margins-are","title":"Distribution of Classification Margins: Are All Data Equal?","date":"2021-07-21","arxiv_id":"2107.10199","n_code_links":0,"syntology":null},{"paper":null,"slug":"improved-learning-rates-for-stochastic","title":"Improved Learning Rates for Stochastic Optimization: Two Theoretical Viewpoints","date":"2021-07-19","arxiv_id":"2107.08686","n_code_links":0,"syntology":null},{"paper":"/paper/non-asymptotic-estimates-for-tusla-algorithm","slug":"non-asymptotic-estimates-for-tusla-algorithm","title":"Non-asymptotic estimates for TUSLA algorithm for non-convex learning with applications to neural networks with ReLU activation function","date":"2021-07-19","arxiv_id":"2107.08649","n_code_links":1,"syntology":null},{"paper":"/paper/rethinking-the-limiting-dynamics-of-sgd","slug":"rethinking-the-limiting-dynamics-of-sgd","title":"The Limiting Dynamics of SGD: Modified Loss, Phase Space Oscillations, and Anomalous Diffusion","date":"2021-07-19","arxiv_id":"2107.09133","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-new-adaptive-gradient-method-with-gradient","title":"A New Adaptive Gradient Method with Gradient Decomposition","date":"2021-07-18","arxiv_id":"2107.08377","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-distributed-k-fac-with-smart","title":"Accelerating Distributed K-FAC with Smart Parallelism of Computing and Communication Tasks","date":"2021-07-14","arxiv_id":"2107.06533","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-the-role-of-implicit-regularization-batch","title":"SGD: The Role of Implicit Regularization, Batch-size and Multiple-epochs","date":"2021-07-11","arxiv_id":"2107.05074","n_code_links":0,"syntology":null}],"record_sha256":"d266c5462874cd120eb7bc0a05f4d17127f45c05090dd215b56483f31c156bbf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}