{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/14","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":14,"pages_in_order":21,"rows_per_page":100,"rows":[1301,1400],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/13","next":"/method/sgd/papers/15","papers":[{"paper":null,"slug":"quantitative-propagation-of-chaos-for-sgd-in","title":"Quantitative Propagation of Chaos for SGD in Wide Neural Networks","date":"2020-07-13","arxiv_id":"2007.06352","n_code_links":0,"syntology":null},{"paper":"/paper/adascale-sgd-a-user-friendly-algorithm-for","slug":"adascale-sgd-a-user-friendly-algorithm-for","title":"AdaScale SGD: A User-Friendly Algorithm for Distributed Training","date":"2020-07-09","arxiv_id":"2007.05105","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":null}},{"paper":null,"slug":"how-benign-is-benign-overfitting","title":"How benign is benign overfitting?","date":"2020-07-08","arxiv_id":"2007.04028","n_code_links":0,"syntology":null},{"paper":null,"slug":"bypassing-the-ambient-dimension-private-sgd","title":"Bypassing the Ambient Dimension: Private SGD with Gradient Subspace Identification","date":"2020-07-07","arxiv_id":"2007.03813","n_code_links":0,"syntology":null},{"paper":null,"slug":"gradient-descent-converges-to-ridgelet","title":"Ridge Regression with Over-Parametrized Two-Layer Networks Converge to Ridgelet Spectrum","date":"2020-07-07","arxiv_id":"2007.03441","n_code_links":0,"syntology":null},{"paper":"/paper/lossless-cnn-channel-pruning-via-gradient","slug":"lossless-cnn-channel-pruning-via-gradient","title":"ResRep: Lossless CNN Pruning via Decoupling Remembering and Forgetting","date":"2020-07-07","arxiv_id":"2007.03260","n_code_links":6,"syntology":null},{"paper":null,"slug":"streaming-complexity-of-svms","title":"Streaming Complexity of SVMs","date":"2020-07-07","arxiv_id":"2007.03633","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-the-impact-of-model-incoherence","title":"Understanding the Impact of Model Incoherence on Convergence of Incremental SGD with Random Reshuffle","date":"2020-07-07","arxiv_id":"2007.03509","n_code_links":0,"syntology":null},{"paper":"/paper/tdprop-does-jacobi-preconditioning-help","slug":"tdprop-does-jacobi-preconditioning-help","title":"TDprop: Does Jacobi Preconditioning Help Temporal Difference Learning?","date":"2020-07-06","arxiv_id":"2007.02786","n_code_links":0,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":null}},{"paper":null,"slug":"weak-error-analysis-for-stochastic-gradient","title":"Weak error analysis for stochastic gradient descent optimization algorithms","date":"2020-07-03","arxiv_id":"2007.02723","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-braking-for-mitigating-gradient","title":"Adaptive Braking for Mitigating Gradient Delay","date":"2020-07-02","arxiv_id":"2007.01397","n_code_links":0,"syntology":null},{"paper":null,"slug":"balancing-rates-and-variance-via-adaptive","title":"Balancing Rates and Variance via Adaptive Batch-Size for Stochastic Optimization Problems","date":"2020-07-02","arxiv_id":"2007.01219","n_code_links":0,"syntology":null},{"paper":"/paper/ecpe-2d-emotion-cause-pair-extraction-based","slug":"ecpe-2d-emotion-cause-pair-extraction-based","title":"ECPE-2D: Emotion-Cause Pair Extraction based on Joint Two-Dimensional Representation, Interaction and Prediction","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"online-robust-regression-via-sgd-on-the-l1","title":"Online Robust Regression via SGD on the l1 loss","date":"2020-07-01","arxiv_id":"2007.00399","n_code_links":0,"syntology":null},{"paper":null,"slug":"adasgd-bridging-the-gap-between-sgd-and-adam","title":"AdaSGD: Bridging the gap between SGD and Adam","date":"2020-06-30","arxiv_id":"2006.16541","n_code_links":0,"syntology":null},{"paper":"/paper/adai-separating-the-effects-of-adaptive","slug":"adai-separating-the-effects-of-adaptive","title":"Adai: Separating the Effects of Adaptive Learning Rate and Momentum Inertia","date":"2020-06-29","arxiv_id":"2006.15815","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zeke-xie/adaptive-inertia-adai"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"understanding-gradient-clipping-in-private","title":"Understanding Gradient Clipping in Private SGD: A Geometric Perspective","date":"2020-06-27","arxiv_id":"2006.15429","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-sgd-a-bayesian-sampler-well-almost","title":"Is SGD a Bayesian sampler? Well, almost","date":"2020-06-26","arxiv_id":"2006.15191","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-generalization-benefit-of-noise-in","title":"On the Generalization Benefit of Noise in Stochastic Gradient Descent","date":"2020-06-26","arxiv_id":"2006.15081","n_code_links":0,"syntology":null},{"paper":null,"slug":"stability-enhanced-privacy-and-applications","title":"Stability Enhanced Privacy and Applications in Private Stochastic Gradient Descent","date":"2020-06-25","arxiv_id":"2006.14360","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-of-stochastic-gradient-descent-with","title":"Dynamic of Stochastic Gradient Descent with State-Dependent Noise","date":"2020-06-24","arxiv_id":"2006.13719","n_code_links":0,"syntology":null},{"paper":null,"slug":"private-stochastic-non-convex-optimization","title":"Private Stochastic Non-Convex Optimization: Adaptive Algorithms and Tighter Generalization Bounds","date":"2020-06-24","arxiv_id":"2006.13501","n_code_links":0,"syntology":null},{"paper":"/paper/spherical-perspective-on-learning-with-batch","slug":"spherical-perspective-on-learning-with-batch","title":"Spherical Perspective on Learning with Normalization Layers","date":"2020-06-23","arxiv_id":"2006.13382","n_code_links":1,"syntology":null},{"paper":null,"slug":"byzantine-resilient-high-dimensional-sgd-with","title":"Byzantine-Resilient High-Dimensional Federated Learning","date":"2020-06-22","arxiv_id":"2006.13041","n_code_links":0,"syntology":null},{"paper":"/paper/adaptive-learning-rates-with-maximum","slug":"adaptive-learning-rates-with-maximum","title":"MaxVA: Fast Adaptation of Step Sizes by Maximizing Observed Variance of Gradients","date":"2020-06-21","arxiv_id":"2006.11918","n_code_links":1,"syntology":null},{"paper":"/paper/generalisation-guarantees-for-continual","slug":"generalisation-guarantees-for-continual","title":"Generalisation Guarantees for Continual Learning with Orthogonal Gradient Descent","date":"2020-06-21","arxiv_id":"2006.11942","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":4,"n_instrument":1,"unverified":4,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["MehdiAbbanaBennani/continual-learning-ogdplus"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"how-do-sgd-hyperparameters-in-natural","title":"How do SGD hyperparameters in natural training affect adversarial robustness?","date":"2020-06-20","arxiv_id":"2006.11604","n_code_links":0,"syntology":null},{"paper":null,"slug":"training-overparametrized-neural-networks-in","title":"Training (Overparametrized) Neural Networks in Near-Linear Time","date":"2020-06-20","arxiv_id":"2006.11648","n_code_links":0,"syntology":null},{"paper":null,"slug":"unified-analysis-of-stochastic-gradient","title":"Unified Analysis of Stochastic Gradient Methods for Composite Convex and Smooth Optimization","date":"2020-06-20","arxiv_id":"2006.11573","n_code_links":0,"syntology":null},{"paper":null,"slug":"deed-a-general-quantization-scheme-for","title":"DEED: A General Quantization Scheme for Communication Efficiency in Bits","date":"2020-06-19","arxiv_id":"2006.11401","n_code_links":0,"syntology":null},{"paper":null,"slug":"differentially-private-variational","title":"Differentially Private Variational Autoencoders with Term-wise Gradient Aggregation","date":"2020-06-19","arxiv_id":"2006.11204","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-almost-sure-convergence-of-stochastic","title":"On the Almost Sure Convergence of Stochastic Gradient Descent in Non-Convex Problems","date":"2020-06-19","arxiv_id":"2006.11144","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-for-structured-nonconvex-functions","title":"SGD for Structured Nonconvex Functions: Learning Rates, Minibatching and Interpolation","date":"2020-06-18","arxiv_id":"2006.10311","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-in-hilbert-scales","title":"Stochastic Gradient Descent in Hilbert Scales: Smoothness, Preconditioning and Earlier Stopping","date":"2020-06-18","arxiv_id":"2006.10840","n_code_links":0,"syntology":null},{"paper":null,"slug":"communication-efficient-robust-federated","title":"Communication-Efficient Robust Federated Learning Over Heterogeneous Datasets","date":"2020-06-17","arxiv_id":"2006.09992","n_code_links":0,"syntology":null},{"paper":"/paper/curvature-is-key-sub-sampled-loss-surfaces","slug":"curvature-is-key-sub-sampled-loss-surfaces","title":"Learning Rates as a Function of Batch Size: A Random Matrix Theory Approach to Neural Network Training","date":"2020-06-16","arxiv_id":"2006.09092","n_code_links":1,"syntology":null},{"paper":"/paper/directional-pruning-of-deep-neural-networks","slug":"directional-pruning-of-deep-neural-networks","title":"Directional Pruning of Deep Neural Networks","date":"2020-06-16","arxiv_id":"2006.09358","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["donlan2710/gRDA-Optimizer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/federated-accelerated-stochastic-gradient","slug":"federated-accelerated-stochastic-gradient","title":"Federated Accelerated Stochastic Gradient Descent","date":"2020-06-16","arxiv_id":"2006.08950","n_code_links":1,"syntology":null},{"paper":null,"slug":"flatness-is-a-false-friend","title":"Flatness is a False Friend","date":"2020-06-16","arxiv_id":"2006.09091","n_code_links":0,"syntology":null},{"paper":"/paper/hausdorff-dimension-stochastic-differential","slug":"hausdorff-dimension-stochastic-differential","title":"Hausdorff Dimension, Heavy Tails, and Generalization in Neural Networks","date":"2020-06-16","arxiv_id":"2006.09313","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["umutsimsekli/Hausdorff-Dimension-and-Generalization"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-sparse-connectivity-adversarial-robustness","title":"On sparse connectivity, adversarial robustness, and a novel model of the artificial neuron","date":"2020-06-16","arxiv_id":"2006.09510","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-grained-analysis-of-stability-and","title":"Fine-Grained Analysis of Stability and Generalization for Stochastic Gradient Descent","date":"2020-06-15","arxiv_id":"2006.08157","n_code_links":0,"syntology":null},{"paper":"/paper/shape-matters-understanding-the-implicit-bias","slug":"shape-matters-understanding-the-implicit-bias","title":"Shape Matters: Understanding the Implicit Bias of the Noise Covariance","date":"2020-06-15","arxiv_id":"2006.08680","n_code_links":1,"syntology":null},{"paper":"/paper/slowing-down-the-weight-norm-increase-in","slug":"slowing-down-the-weight-norm-increase-in","title":"AdamP: Slowing Down the Slowdown for Momentum Optimizers on Scale-invariant Weights","date":"2020-06-15","arxiv_id":"2006.08217","n_code_links":4,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["clovaai/AdamP"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"paper":null,"slug":"spherical-motion-dynamics-of-deep-neural","title":"Spherical Motion Dynamics: Learning Dynamics of Neural Network with Normalization, Weight Decay, and SGD","date":"2020-06-15","arxiv_id":"2006.08419","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-analysis-of-constant-step-size-sgd-in-the","title":"An Analysis of Constant Step Size SGD in the Non-convex Regime: Asymptotic Normality and Bias","date":"2020-06-14","arxiv_id":"2006.07904","n_code_links":0,"syntology":null},{"paper":null,"slug":"differentially-private-decentralized-learning","title":"Topology-aware Differential Privacy for Decentralized Image Classification","date":"2020-06-14","arxiv_id":"2006.07817","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-the-stochastic-heavy","title":"Almost sure convergence rates for Stochastic Gradient Descent and Stochastic Heavy Ball","date":"2020-06-14","arxiv_id":"2006.07867","n_code_links":0,"syntology":null},{"paper":"/paper/auditing-differentially-private-machine","slug":"auditing-differentially-private-machine","title":"Auditing Differentially Private Machine Learning: How Private is Private SGD?","date":"2020-06-13","arxiv_id":"2006.07709","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jagielski/auditing-dpsgd"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-pitfalls-of-simplicity-bias-in-neural","slug":"the-pitfalls-of-simplicity-bias-in-neural","title":"The Pitfalls of Simplicity Bias in Neural Networks","date":"2020-06-13","arxiv_id":"2006.07710","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-unified-analysis-of-stochastic-gradient","title":"A Unified Analysis of Stochastic Gradient Methods for Nonconvex Federated Optimization","date":"2020-06-12","arxiv_id":"2006.07013","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-gradient-methods-can-be-provably","title":"Adaptive Gradient Methods Can Be Provably Faster than SGD after Finite Epochs","date":"2020-06-12","arxiv_id":"2006.07037","n_code_links":0,"syntology":null},{"paper":null,"slug":"o-1-communication-for-distributed-sgd-through","title":"O(1) Communication for Distributed SGD through Two-Level Gradient Averaging","date":"2020-06-12","arxiv_id":"2006.07405","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-with-shuffling-optimal-rates-without","title":"SGD with shuffling: optimal rates without component convexity and large epoch requirements","date":"2020-06-12","arxiv_id":"2006.06946","n_code_links":0,"syntology":null},{"paper":null,"slug":"stability-of-stochastic-gradient-descent-on","title":"Stability of Stochastic Gradient Descent on Nonsmooth Convex Losses","date":"2020-06-12","arxiv_id":"2006.06914","n_code_links":0,"syntology":null},{"paper":"/paper/adaptive-gradient-methods-converge-faster","slug":"adaptive-gradient-methods-converge-faster","title":"Adaptive Gradient Methods Converge Faster with Over-Parameterization (but you should do a line-search)","date":"2020-06-11","arxiv_id":"2006.06835","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/adas-adaptive-scheduling-of-stochastic","slug":"adas-adaptive-scheduling-of-stochastic","title":"AdaS: Adaptive Scheduling of Stochastic Gradients","date":"2020-06-11","arxiv_id":"2006.06587","n_code_links":2,"syntology":null},{"paper":null,"slug":"borrowing-from-the-future-addressing-double","title":"Borrowing From the Future: Addressing Double Sampling in Model-free Control","date":"2020-06-11","arxiv_id":"2006.06173","n_code_links":0,"syntology":null},{"paper":null,"slug":"multiplicative-noise-and-heavy-tails-in","title":"Multiplicative noise and heavy tails in stochastic optimization","date":"2020-06-11","arxiv_id":"2006.06293","n_code_links":0,"syntology":null},{"paper":null,"slug":"non-convex-sgd-learns-halfspaces-with","title":"Non-Convex SGD Learns Halfspaces with Adversarial Label Noise","date":"2020-06-11","arxiv_id":"2006.06742","n_code_links":0,"syntology":null},{"paper":null,"slug":"stl-sgd-speeding-up-local-sgd-with-stagewise","title":"STL-SGD: Speeding Up Local SGD with Stagewise Communication Period","date":"2020-06-11","arxiv_id":"2006.06377","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-regularisation-methods-for","slug":"understanding-regularisation-methods-for","title":"Unifying Regularisation Methods for Continual Learning","date":"2020-06-11","arxiv_id":"2006.06357","n_code_links":2,"syntology":null},{"paper":null,"slug":"dynamical-mean-field-theory-for-stochastic","title":"Dynamical mean-field theory for stochastic gradient descent in Gaussian mixture classification","date":"2020-06-10","arxiv_id":"2006.06098","n_code_links":0,"syntology":null},{"paper":"/paper/random-reshuffling-simple-analysis-with-vast","slug":"random-reshuffling-simple-analysis-with-vast","title":"Random Reshuffling: Simple Analysis with Vast Improvements","date":"2020-06-10","arxiv_id":"2006.05988","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["konstmish/random_reshuffling"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/sketchy-empirical-natural-gradient-methods","slug":"sketchy-empirical-natural-gradient-methods","title":"Sketchy Empirical Natural Gradient Methods for Deep Learning","date":"2020-06-10","arxiv_id":"2006.05924","n_code_links":1,"syntology":null},{"paper":null,"slug":"minibatch-vs-local-sgd-for-heterogeneous","title":"Minibatch vs Local SGD for Heterogeneous Distributed Learning","date":"2020-06-08","arxiv_id":"2006.04735","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-optimization-with-non-stationary","title":"Beyond Worst-Case Analysis in Stochastic Approximation: Moment Estimation Improves Instance Complexity","date":"2020-06-08","arxiv_id":"2006.04429","n_code_links":0,"syntology":null},{"paper":"/paper/the-heavy-tail-phenomenon-in-sgd","slug":"the-heavy-tail-phenomenon-in-sgd","title":"The Heavy-Tail Phenomenon in SGD","date":"2020-06-08","arxiv_id":"2006.04740","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-strength-of-nesterov-s-extrapolation-in","title":"The Strength of Nesterov's Extrapolation in the Individual Convergence of Nonsmooth Optimization","date":"2020-06-08","arxiv_id":"2006.04340","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-efficient-algorithm-for-generalized-linear","title":"An Efficient Algorithm For Generalized Linear Bandit: Online Stochastic Gradient Descent and Thompson Sampling","date":"2020-06-07","arxiv_id":"2006.04012","n_code_links":0,"syntology":null},{"paper":null,"slug":"bayesian-neural-network-via-stochastic","title":"Bayesian Neural Network via Stochastic Gradient Descent","date":"2020-06-04","arxiv_id":"2006.08453","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-distributed-training-with-adaptive","title":"Scaling Distributed Training with Adaptive Summation","date":"2020-06-04","arxiv_id":"2006.02924","n_code_links":0,"syntology":null},{"paper":"/paper/towards-asymptotic-optimality-with","slug":"towards-asymptotic-optimality-with","title":"Asymptotic Analysis of Conditioned Stochastic Gradient Descent","date":"2020-06-04","arxiv_id":"2006.02745","n_code_links":1,"syntology":null},{"paper":null,"slug":"local-sgd-with-a-communication-overhead","title":"Local SGD With a Communication Overhead Depending Only on the Number of Workers","date":"2020-06-03","arxiv_id":"2006.02582","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-promise-of-the-stochastic-generalized","slug":"on-the-promise-of-the-stochastic-generalized","title":"On the Promise of the Stochastic Generalized Gauss-Newton Method for Training DNNs","date":"2020-06-03","arxiv_id":"2006.02409","n_code_links":2,"syntology":null},{"paper":"/paper/adahessian-an-adaptive-second-order-optimizer","slug":"adahessian-an-adaptive-second-order-optimizer","title":"ADAHESSIAN: An Adaptive Second Order Optimizer for Machine Learning","date":"2020-06-01","arxiv_id":"2006.00719","n_code_links":4,"syntology":{"ran":5,"of":11,"n_ran_checked":4,"n_instrument":1,"unverified":6,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["amirgholami/adahessian"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"augment-your-batch-improving-generalization","title":"Augment Your Batch: Improving Generalization Through Instance Repetition","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"auto-tuning-structured-light-by-optical","title":"Auto-Tuning Structured Light by Optical Stochastic Gradient Descent","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"dasgd-squeezing-sgd-parallelization","title":"DaSGD: Squeezing SGD Parallelization Performance in Distributed Training Using Delayed Averaging","date":"2020-05-31","arxiv_id":"2006.00441","n_code_links":0,"syntology":null},{"paper":null,"slug":"inherent-noise-in-gradient-based-methods","title":"Inherent Noise in Gradient Based Methods","date":"2020-05-26","arxiv_id":"2005.12743","n_code_links":0,"syntology":null},{"paper":null,"slug":"microphone-array-based-surveillance-audio","title":"Microphone Array Based Surveillance Audio Classification","date":"2020-05-22","arxiv_id":"2005.11348","n_code_links":0,"syntology":null},{"paper":"/paper/accelerated-convergence-for-counterfactual","slug":"accelerated-convergence-for-counterfactual","title":"Accelerated Convergence for Counterfactual Learning to Rank","date":"2020-05-21","arxiv_id":"2005.10615","n_code_links":1,"syntology":null},{"paper":null,"slug":"rtop-k-a-statistical-estimation-approach-to","title":"rTop-k: A Statistical Estimation Approach to Distributed SGD","date":"2020-05-21","arxiv_id":"2005.10761","n_code_links":0,"syntology":null},{"paper":"/paper/stochastic-optimization-with-heavy-tailed","slug":"stochastic-optimization-with-heavy-tailed","title":"Stochastic Optimization with Heavy-Tailed Noise via Accelerated Gradient Clipping","date":"2020-05-21","arxiv_id":"2005.10785","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":0,"n_instrument":2,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["eduardgorbunov/accelerated_clipping"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"byzantine-resilient-sgd-in-high-dimensions-on","title":"Byzantine-Resilient SGD in High Dimensions on Heterogeneous Data","date":"2020-05-16","arxiv_id":"2005.07866","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-the-gravitational-force-law-and","title":"Learning the gravitational force law and other analytic functions","date":"2020-05-15","arxiv_id":"2005.07724","n_code_links":0,"syntology":null},{"paper":"/paper/od-sgd-one-step-delay-stochastic-gradient","slug":"od-sgd-one-step-delay-stochastic-gradient","title":"OD-SGD: One-step Delay Stochastic Gradient Descent for Distributed Training","date":"2020-05-14","arxiv_id":"2005.06728","n_code_links":1,"syntology":null},{"paper":null,"slug":"squarm-sgd-communication-efficient-momentum","title":"SQuARM-SGD: Communication-Efficient Momentum SGD for Decentralized Optimization","date":"2020-05-13","arxiv_id":"2005.07041","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-of-online-adaptive-and-recurrent","title":"Convergence of Online Adaptive and Recurrent Optimization Algorithms","date":"2020-05-12","arxiv_id":"2005.05645","n_code_links":0,"syntology":null},{"paper":null,"slug":"rso-a-gradient-free-sampling-based-approach","title":"RSO: A Gradient Free Sampling Based Approach For Training Deep Neural Networks","date":"2020-05-12","arxiv_id":"2005.05955","n_code_links":0,"syntology":null},{"paper":"/paper/geoopt-riemannian-optimization-in-pytorch","slug":"geoopt-riemannian-optimization-in-pytorch","title":"Geoopt: Riemannian Optimization in PyTorch","date":"2020-05-06","arxiv_id":"2005.02819","n_code_links":2,"syntology":null},{"paper":null,"slug":"adaptive-learning-of-the-optimal-mini-batch","title":"Adaptive Learning of the Optimal Batch Size of SGD","date":"2020-05-03","arxiv_id":"2005.01097","n_code_links":0,"syntology":null},{"paper":null,"slug":"riemannian-stochastic-proximal-gradient","title":"Riemannian Stochastic Proximal Gradient Methods for Nonsmooth Optimization over the Stiefel Manifold","date":"2020-05-03","arxiv_id":"2005.01209","n_code_links":0,"syntology":null},{"paper":null,"slug":"gap-aware-mitigation-of-gradient-staleness-1","title":"Gap-Aware Mitigation of Gradient Staleness","date":"2020-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"breaking-global-barriers-in-parallel","title":"Breaking (Global) Barriers in Parallel Stochastic Optimization with Wait-Avoiding Group Averaging","date":"2020-04-30","arxiv_id":"2005.00124","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-polynomials-of-few-relevant","title":"Learning Polynomials of Few Relevant Dimensions","date":"2020-04-28","arxiv_id":"2004.13748","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-impact-of-the-mini-batch-size-on-the","title":"The Impact of the Mini-batch Size on the Variance of Gradients in Stochastic Gradient Descent","date":"2020-04-27","arxiv_id":"2004.13146","n_code_links":0,"syntology":null},{"paper":"/paper/federated-learning-with-only-positive-labels","slug":"federated-learning-with-only-positive-labels","title":"Federated Learning with Only Positive Labels","date":"2020-04-21","arxiv_id":"2004.10342","n_code_links":1,"syntology":null},{"paper":null,"slug":"stochastic-gradient-algorithms-from-ode","title":"Stochastic gradient algorithms from ODE splitting perspective","date":"2020-04-19","arxiv_id":"2004.08981","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-tight-convergence-rates-of-without","title":"On Tight Convergence Rates of Without-replacement SGD","date":"2020-04-18","arxiv_id":"2004.08657","n_code_links":0,"syntology":null}],"record_sha256":"c5da2095f92f11e81041930fe1ab8ea14cc902420c06d21c25d85a76effdd5c6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}