{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/12","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":12,"pages_in_order":21,"rows_per_page":100,"rows":[1101,1200],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/11","next":"/method/sgd/papers/13","papers":[{"paper":null,"slug":"can-single-shuffle-sgd-be-better-than","title":"Can Single-Shuffle SGD be Better than Reshuffling SGD and GD?","date":"2021-03-12","arxiv_id":"2103.07079","n_code_links":0,"syntology":null},{"paper":null,"slug":"streaming-linear-system-identification-with","title":"Streaming Linear System Identification with Reverse Experience Replay","date":"2021-03-10","arxiv_id":"2103.05896","n_code_links":0,"syntology":null},{"paper":null,"slug":"why-flatness-correlates-with-generalization","title":"Why flatness does and does not correlate with generalization for deep neural networks","date":"2021-03-10","arxiv_id":"2103.06219","n_code_links":0,"syntology":null},{"paper":null,"slug":"escaping-saddle-points-with-stochastically","title":"Escaping Saddle Points with Stochastically Controlled Stochastic Gradient Methods","date":"2021-03-07","arxiv_id":"2103.04413","n_code_links":0,"syntology":null},{"paper":"/paper/second-order-step-size-tuning-of-sgd-for-non","slug":"second-order-step-size-tuning-of-sgd-for-non","title":"Second-order step-size tuning of SGD for non-convex optimization","date":"2021-03-05","arxiv_id":"2103.03570","n_code_links":1,"syntology":null},{"paper":"/paper/correcting-momentum-with-second-order","slug":"correcting-momentum-with-second-order","title":"Better SGD using Second-order Momentum","date":"2021-03-04","arxiv_id":"2103.03265","n_code_links":1,"syntology":null},{"paper":null,"slug":"pac-bayes-and-information-complexity","title":"PAC-Bayes and Information Complexity","date":"2021-03-04","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/a-biased-graph-neural-network-sampler-with","slug":"a-biased-graph-neural-network-sampler-with","title":"A Biased Graph Neural Network Sampler with Near-Optimal Regret","date":"2021-03-01","arxiv_id":"2103.01089","n_code_links":1,"syntology":null},{"paper":null,"slug":"non-euclidean-differentially-private","title":"Non-Euclidean Differentially Private Stochastic Convex Optimization: Optimal Rates in Linear Time","date":"2021-03-01","arxiv_id":"2103.01278","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-neural-networks-with-relu-sine","title":"Deep Neural Networks with ReLU-Sine-Exponential Activations Break Curse of Dimensionality in Approximation on Hölder Class","date":"2021-02-28","arxiv_id":"2103.00542","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-utility-of-gradient-compression-in","slug":"on-the-utility-of-gradient-compression-in","title":"On the Utility of Gradient Compression in Distributed Training Systems","date":"2021-02-28","arxiv_id":"2103.00543","n_code_links":1,"syntology":null},{"paper":null,"slug":"experiments-with-rich-regime-training-for","title":"Experiments with Rich Regime Training for Deep Learning","date":"2021-02-26","arxiv_id":"2102.13522","n_code_links":0,"syntology":null},{"paper":null,"slug":"noisy-truncated-sgd-optimization-and","title":"Noisy Truncated SGD: Optimization and Generalization","date":"2021-02-26","arxiv_id":"2103.00075","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-generalization-of-stochastic-gradient","title":"On the Generalization of Stochastic Gradient Descent with Momentum","date":"2021-02-26","arxiv_id":"2102.13653","n_code_links":0,"syntology":null},{"paper":null,"slug":"local-stochastic-gradient-descent-ascent","title":"Local Stochastic Gradient Descent Ascent: Convergence Analysis and Communication Efficiency","date":"2021-02-25","arxiv_id":"2102.13152","n_code_links":0,"syntology":null},{"paper":"/paper/loss-surface-simplexes-for-mode-connecting","slug":"loss-surface-simplexes-for-mode-connecting","title":"Loss Surface Simplexes for Mode Connecting Volumes and Fast Ensembling","date":"2021-02-25","arxiv_id":"2102.13042","n_code_links":1,"syntology":null},{"paper":null,"slug":"machine-unlearning-via-algorithmic-stability","title":"Machine Unlearning via Algorithmic Stability","date":"2021-02-25","arxiv_id":"2102.13179","n_code_links":0,"syntology":null},{"paper":null,"slug":"mixed-variable-bayesian-optimization-with","title":"Mixed Variable Bayesian Optimization with Frequency Modulated Kernels","date":"2021-02-25","arxiv_id":"2102.12792","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-validity-of-modeling-sgd-with","slug":"on-the-validity-of-modeling-sgd-with","title":"On the Validity of Modeling SGD with Stochastic Differential Equations (SDEs)","date":"2021-02-24","arxiv_id":"2102.12470","n_code_links":1,"syntology":null},{"paper":null,"slug":"online-stochastic-gradient-descent-learns","title":"Online Stochastic Gradient Descent Learns Linear Dynamical Systems from A Single Trajectory","date":"2021-02-23","arxiv_id":"2102.11822","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-rates-of-stochastic-gradient","title":"Convergence Rates of Stochastic Gradient Descent under Infinite Noise Variance","date":"2021-02-20","arxiv_id":"2102.10346","n_code_links":0,"syntology":null},{"paper":"/paper/permutation-based-sgd-is-random-optimal","slug":"permutation-based-sgd-is-random-optimal","title":"Permutation-Based SGD: Is Random Optimal?","date":"2021-02-19","arxiv_id":"2102.09718","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 4 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["shashankrajput/flipflop"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/personalized-federated-learning-a-unified","slug":"personalized-federated-learning-a-unified","title":"Personalized Federated Learning: A Unified Framework and Universal Optimization Techniques","date":"2021-02-19","arxiv_id":"2102.09743","n_code_links":1,"syntology":null},{"paper":null,"slug":"svrg-meets-adagrad-painless-variance","title":"SVRG Meets AdaGrad: Painless Variance Reduction","date":"2021-02-18","arxiv_id":"2102.09645","n_code_links":0,"syntology":null},{"paper":null,"slug":"differential-private-hogwild-over-distributed","title":"Proactive DP: A Multple Target Optimization Framework for DP-SGD","date":"2021-02-17","arxiv_id":"2102.09030","n_code_links":0,"syntology":null},{"paper":null,"slug":"genetically-optimized-prediction-of-remaining","title":"Genetically Optimized Prediction of Remaining Useful Life","date":"2021-02-17","arxiv_id":"2102.08845","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-of-stochastic-gradient-descent-1","title":"Convergence of stochastic gradient descent schemes for Lojasiewicz-landscapes","date":"2021-02-16","arxiv_id":"2102.09385","n_code_links":0,"syntology":null},{"paper":"/paper/differential-privacy-and-byzantine-resilience","slug":"differential-privacy-and-byzantine-resilience","title":"Differential Privacy and Byzantine Resilience in SGD: Do They Add Up?","date":"2021-02-16","arxiv_id":"2102.08166","n_code_links":1,"syntology":null},{"paper":"/paper/gradinit-learning-to-initialize-neural","slug":"gradinit-learning-to-initialize-neural","title":"GradInit: Learning to Initialize Neural Networks for Stable and Efficient Training","date":"2021-02-16","arxiv_id":"2102.08098","n_code_links":2,"syntology":{"ran":7,"of":13,"n_ran_checked":4,"n_instrument":3,"unverified":6,"pointer_only":12,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","official":{"repos":["zhuchen03/gradinit"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/intsgd-floatless-compression-of-stochastic","slug":"intsgd-floatless-compression-of-stochastic","title":"IntSGD: Adaptive Floatless Compression of Stochastic Gradients","date":"2021-02-16","arxiv_id":"2102.08374","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":1,"n_instrument":1,"unverified":3,"pointer_only":5,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["bokunwang1/intsgd"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/does-standard-backpropagation-forget-less","slug":"does-standard-backpropagation-forget-less","title":"Does the Adam Optimizer Exacerbate Catastrophic Forgetting?","date":"2021-02-15","arxiv_id":"2102.07686","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-role-of-momentum-parameters-in-the-1","title":"The Role of Momentum Parameters in the Optimal Convergence of Adaptive Polyak's Heavy-ball Methods","date":"2021-02-15","arxiv_id":"2102.07314","n_code_links":0,"syntology":null},{"paper":"/paper/learning-by-turning-neural-architecture-aware","slug":"learning-by-turning-neural-architecture-aware","title":"Learning by Turning: Neural Architecture Aware Optimisation","date":"2021-02-14","arxiv_id":"2102.07227","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":2,"n_instrument":3,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jxbz/nero"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/asymmetric-heavy-tails-and-implicit-bias-in","slug":"asymmetric-heavy-tails-and-implicit-bias-in","title":"Asymmetric Heavy Tails and Implicit Bias in Gaussian Noise Injections","date":"2021-02-13","arxiv_id":"2102.07006","n_code_links":1,"syntology":null},{"paper":"/paper/bayesian-neural-network-priors-revisited","slug":"bayesian-neural-network-priors-revisited","title":"Bayesian Neural Network Priors Revisited","date":"2021-02-12","arxiv_id":"2102.06571","n_code_links":1,"syntology":null},{"paper":"/paper/proximal-and-federated-random-reshuffling","slug":"proximal-and-federated-random-reshuffling","title":"Proximal and Federated Random Reshuffling","date":"2021-02-12","arxiv_id":"2102.06704","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["konstmish/rr_prox_fed"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"stability-and-convergence-of-stochastic","title":"Stability and Convergence of Stochastic Gradient Clipping: Beyond Lipschitz Continuity and Smoothness","date":"2021-02-12","arxiv_id":"2102.06489","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-minibatch-noise-discrete-time-sgd","title":"Strength of Minibatch Noise in SGD","date":"2021-02-10","arxiv_id":"2102.05375","n_code_links":0,"syntology":null},{"paper":"/paper/on-pytorch-implementation-of-density","slug":"on-pytorch-implementation-of-density","title":"On PyTorch Implementation of Density Estimators for von Mises-Fisher and Its Mixture","date":"2021-02-10","arxiv_id":"2102.05340","n_code_links":1,"syntology":null},{"paper":null,"slug":"federated-learning-with-local-differential","title":"Federated Learning with Local Differential Privacy: Trade-offs between Privacy, Utility, and Communication","date":"2021-02-09","arxiv_id":"2102.04737","n_code_links":0,"syntology":null},{"paper":null,"slug":"double-momentum-sgd-for-federated-learning","title":"Coordinating Momenta for Cross-silo Federated Learning","date":"2021-02-08","arxiv_id":"2102.03970","n_code_links":0,"syntology":null},{"paper":null,"slug":"eliminating-sharp-minima-from-sgd-with","title":"Eliminating Sharp Minima from SGD with Truncated Heavy-tailed Noise","date":"2021-02-08","arxiv_id":"2102.04297","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-in-the-large-average-case-analysis","title":"SGD in the Large: Average-case Analysis, Asymptotics, and Stepsize Criticality","date":"2021-02-08","arxiv_id":"2102.04396","n_code_links":0,"syntology":null},{"paper":null,"slug":"dimension-free-generalization-bounds-for-non","title":"Dimension Free Generalization Bounds for Non Linear Metric Learning","date":"2021-02-07","arxiv_id":"2102.03802","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-implicit-biases-of-stochastic-gradient","title":"Weight Rescaling: Effective and Robust Regularization for Deep Neural Networks with Batch Normalization","date":"2021-02-06","arxiv_id":"2102.03497","n_code_links":0,"syntology":null},{"paper":null,"slug":"bias-variance-reduced-local-sgd-for-less","title":"Bias-Variance Reduced Local SGD for Less Heterogeneous Federated Learning","date":"2021-02-05","arxiv_id":"2102.03198","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-and-memory-efficient-differentially","title":"Fast and Memory Efficient Differentially Private-SGD via JL Projections","date":"2021-02-05","arxiv_id":"2102.03013","n_code_links":0,"syntology":null},{"paper":null,"slug":"last-iterate-convergence-of-sgd-for-least","title":"Last iterate convergence of SGD for Least-Squares in the Interpolation regime","date":"2021-02-05","arxiv_id":"2102.03183","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-while-dissipating-information","title":"Generalization Bounds for Noisy Iterative Algorithms Using Properties of Additive Noise Channels","date":"2021-02-05","arxiv_id":"2102.02976","n_code_links":0,"syntology":null},{"paper":"/paper/1-bit-adam-communication-efficient-large","slug":"1-bit-adam-communication-efficient-large","title":"1-bit Adam: Communication Efficient Large-Scale Training with Adam's Convergence Speed","date":"2021-02-04","arxiv_id":"2102.02888","n_code_links":2,"syntology":null},{"paper":null,"slug":"stability-and-generalization-of-the","title":"Stability and Generalization of the Decentralized Stochastic Gradient Descent","date":"2021-02-02","arxiv_id":"2102.01302","n_code_links":0,"syntology":null},{"paper":null,"slug":"information-theoretic-generalization-bounds-1","title":"Information-Theoretic Generalization Bounds for Stochastic Gradient Descent","date":"2021-02-01","arxiv_id":"2102.00931","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-generalizes-better-than-gd-and","title":"SGD Generalizes Better Than GD (And Regularization Doesn't Help)","date":"2021-02-01","arxiv_id":"2102.01117","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-origin-of-implicit-regularization-in-1","title":"On the Origin of Implicit Regularization in Stochastic Gradient Descent","date":"2021-01-28","arxiv_id":"2101.12176","n_code_links":0,"syntology":null},{"paper":"/paper/adaptivity-without-compromise-a-momentumized","slug":"adaptivity-without-compromise-a-momentumized","title":"Adaptivity without Compromise: A Momentumized, Adaptive, Dual Averaged Gradient Method for Stochastic Optimization","date":"2021-01-26","arxiv_id":"2101.11075","n_code_links":5,"syntology":null},{"paper":null,"slug":"differentially-private-sgd-with-non-smooth","title":"Differentially Private SGD with Non-Smooth Losses","date":"2021-01-22","arxiv_id":"2101.08925","n_code_links":0,"syntology":null},{"paper":null,"slug":"clairvoyant-prefetching-for-distributed","title":"Clairvoyant Prefetching for Distributed Machine Learning I/O","date":"2021-01-21","arxiv_id":"2101.08734","n_code_links":0,"syntology":null},{"paper":null,"slug":"rate-region-for-indirect-multiterminal-source","title":"Sum-Rate-Distortion Function for Indirect Multiterminal Source Coding in Federated Learning","date":"2021-01-21","arxiv_id":"2101.08696","n_code_links":0,"syntology":null},{"paper":null,"slug":"guided-parallelized-stochastic-gradient","title":"Guided parallelized stochastic gradient descent for delay compensation","date":"2021-01-17","arxiv_id":"2101.07259","n_code_links":0,"syntology":null},{"paper":null,"slug":"linguistically-enriched-and-context-aware","title":"Linguistically-Enriched and Context-Aware Zero-shot Slot Filling","date":"2021-01-16","arxiv_id":"2101.06514","n_code_links":0,"syntology":null},{"paper":null,"slug":"phases-of-learning-dynamics-in-artificial","title":"Phases of learning dynamics in artificial neural networks: with or without mislabeled data","date":"2021-01-16","arxiv_id":"2101.06509","n_code_links":0,"syntology":null},{"paper":null,"slug":"bn-invariant-sharpness-regularizes-the","title":"BN-invariant sharpness regularizes the training model to better generalization","date":"2021-01-08","arxiv_id":"2101.02944","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-understanding-learning-in-neural","title":"Towards Understanding Learning in Neural Networks with Linear Teachers","date":"2021-01-07","arxiv_id":"2101.02533","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-transfer-neuroevolution-tractably-solve","title":"Can Transfer Neuroevolution Tractably Solve Your Differential Equations?","date":"2021-01-06","arxiv_id":"2101.01998","n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-learning-over-noisy-channels","title":"Federated Learning over Noisy Channels: Convergence Analysis and Design Examples","date":"2021-01-06","arxiv_id":"2101.02198","n_code_links":0,"syntology":null},{"paper":"/paper/provable-generalization-of-sgd-trained-neural","slug":"provable-generalization-of-sgd-trained-neural","title":"Provable Generalization of SGD-trained Neural Networks of Any Width in the Presence of Adversarial Label Noise","date":"2021-01-04","arxiv_id":"2101.01152","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-chaos-theory-approach-to-understand-neural","title":"A Chaos Theory Approach to Understand Neural Network Optimization","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-variant-of-stochastic-heavy-ball","title":"A New Variant of Stochastic Heavy ball Optimization Method for Deep Learning","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-dnn-training-through-selective","title":"Accelerating DNN Training through Selective Localized Learning","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-dataset-sampling-by-deep-policy","title":"Adaptive Dataset Sampling by Deep Policy Gradient","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-gradient-methods-can-be-provably-1","title":"Adaptive Gradient Methods Can Be Provably Faster than SGD with Random Shuffling","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"apollo-an-adaptive-parameter-wised-diagonal","title":"Apollo: An Adaptive Parameter-wised Diagonal Quasi-Newton Method for Nonconvex Stochastic Optimization","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"compressing-gradients-in-distributed-sgd-by","title":"Compressing gradients in distributed SGD by exploiting their temporal correlation","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"constructing-multiple-high-quality-deep","title":"Constructing Multiple High-Quality Deep Neural Networks: A TRUST-TECH Based Approach","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"dqsgd-dynamic-quantized-stochastic-gradient","title":"DQSGD: DYNAMIC QUANTIZED STOCHASTIC GRADIENT DESCENT FOR COMMUNICATION-EFFICIENT DISTRIBUTED LEARNING","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/factor-normalization-for-deep-neural-network","slug":"factor-normalization-for-deep-neural-network","title":"Factor Normalization for Deep Neural Network Models","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"fast-convergence-of-stochastic-subgradient","title":"Fast convergence of stochastic subgradient method under interpolation","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"flatness-is-a-flase-friend","title":"Flatness is a Flase Friend","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"how-benign-is-benign-overfitting-1","title":"How Benign is Benign Overfitting ?","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"implicit-regularization-effects-of-unbiased","title":"Implicit Regularization Effects of Unbiased Random Label Noises with SGD","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"implicit-regularization-of-sgd-via","title":"Implicit Regularization of SGD via Thermophoresis","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"local-sgd-meets-asynchrony","title":"Local SGD Meets Asynchrony","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-level-local-sgd-distributed-sgd-for","title":"Multi-Level Local SGD: Distributed SGD for Heterogeneous Hierarchical Networks","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"noise-against-noise-stochastic-label-noise","title":"Noise against noise: stochastic label noise helps combat inherent label noise","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-inductive-bias-of-a-cnn-for","title":"On the Inductive Bias of a CNN for Distributions with Orthogonal Patterns","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-quantized-neural-networks-with","title":"Optimizing Quantized Neural Networks with Natural Gradient","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-the-stability-of-stochastic","title":"Revisiting the Stability of Stochastic Gradient Descent: A Tightness Analysis","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-optimization-with-non-stationary-1","title":"Stochastic Optimization with Non-stationary Noise: The Power of Moment Estimation","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-proximal-point-algorithm-for-large","title":"Stochastic Proximal Point Algorithm for Large-scale Nonconvex Optimization: Convergence, Implementation, and Application to Neural Networks","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"symmetry-conservation-laws-and-learning","title":"Symmetry, Conservation Laws, and Learning Dynamics in Neural Networks","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-impact-of-the-mini-batch-size-on-the-1","title":"The Impact of the Mini-batch Size on the Dynamics of SGD: Variance and Beyond","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-simpler-the-better-vanilla-sgd-revisited","title":"The simpler the better: vanilla sgd revisited","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"why-does-decentralized-training-outperform","title":"Why Does Decentralized Training Outperform Synchronous Training In The Large Batch Setting?","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-distributional-dynamics-of-three-layer","title":"SGD Distributional Dynamics of Three Layer Neural Networks","date":"2020-12-30","arxiv_id":"2012.15036","n_code_links":0,"syntology":null},{"paper":null,"slug":"catastrophic-fisher-explosion-early-phase-1","title":"Catastrophic Fisher Explosion: Early Phase Fisher Matrix Impacts Generalization","date":"2020-12-28","arxiv_id":"2012.14193","n_code_links":0,"syntology":null},{"paper":null,"slug":"asymptoticng-a-regularized-natural-gradient","title":"AsymptoticNG: A regularized natural gradient optimization algorithm with look-ahead strategy","date":"2020-12-24","arxiv_id":"2012.13077","n_code_links":0,"syntology":null},{"paper":"/paper/training-convolutional-neural-networks-with-3","slug":"training-convolutional-neural-networks-with-3","title":"Training Convolutional Neural Networks With Hebbian Principal Component Analysis","date":"2020-12-22","arxiv_id":"2012.12229","n_code_links":1,"syntology":null},{"paper":null,"slug":"optimizing-deep-neural-networks-through","title":"Optimizing Deep Neural Networks through Neuroevolution with Stochastic Gradient Descent","date":"2020-12-21","arxiv_id":"2012.11184","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-hybrid-mga-msgd-ann-training-approach-for","title":"A hybrid MGA-MSGD ANN training approach for approximate solution of linear elliptic PDEs","date":"2020-12-18","arxiv_id":"2012.11517","n_code_links":0,"syntology":null},{"paper":null,"slug":"fedadc-accelerated-federated-learning-with","title":"FedADC: Accelerated Federated Learning with Drift Control","date":"2020-12-16","arxiv_id":"2012.09102","n_code_links":0,"syntology":null}],"record_sha256":"b1180cae8090e8cb406c4367ce087aee0a09f93c01cd8ad73a54c01458ac69ea","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}