{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/13","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":13,"pages_in_order":21,"rows_per_page":100,"rows":[1201,1300],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/12","next":"/method/sgd/papers/14","papers":[{"paper":"/paper/attentional-biased-stochastic-gradient-for","slug":"attentional-biased-stochastic-gradient-for","title":"Attentional-Biased Stochastic Gradient Descent","date":"2020-12-13","arxiv_id":"2012.06951","n_code_links":1,"syntology":null},{"paper":"/paper/neural-mechanics-symmetry-and-broken","slug":"neural-mechanics-symmetry-and-broken","title":"Neural Mechanics: Symmetry and Broken Conservation Laws in Deep Learning Dynamics","date":"2020-12-08","arxiv_id":"2012.04728","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["danielkunin/neural-mechanics"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"stochastic-gradient-descent-with-large","title":"Noise and Fluctuation of Finite Learning Rate Stochastic Gradient Descent","date":"2020-12-07","arxiv_id":"2012.03636","n_code_links":0,"syntology":null},{"paper":null,"slug":"effect-of-the-initial-configuration-of","title":"Effect of the initial configuration of weights on the training and function of artificial neural networks","date":"2020-12-04","arxiv_id":"2012.02550","n_code_links":0,"syntology":null},{"paper":null,"slug":"lookahead-optimizer-improves-the-performance","title":"Lookahead optimizer improves the performance of Convolutional Autoencoders for reconstruction of natural images","date":"2020-12-03","arxiv_id":"2012.05694","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-with-nonlinear","title":"Stochastic Gradient Descent with Nonlinear Conjugate Gradient-Style Adaptive Momentum","date":"2020-12-03","arxiv_id":"2012.02188","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-and-sample-complexity-of-sgd-in","title":"Convergence and Sample Complexity of SGD in GANs","date":"2020-12-01","arxiv_id":"2012.00732","n_code_links":0,"syntology":null},{"paper":null,"slug":"curriculum-learning-by-dynamic-instance","title":"Curriculum Learning by Dynamic Instance Hardness","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-universality-of-deep-learning","title":"On the universality of deep learning","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robustness-analysis-of-non-convex-stochastic","title":"Robustness Analysis of Non-Convex Stochastic Gradient Descent using Biased Expectations","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-in-correlated","title":"Stochastic Gradient Descent in Correlated Settings: A Study on Gaussian Processes","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-better-generalization-of-adaptive","title":"Towards Better Generalization of Adaptive Gradient Methods","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"eigenvalue-corrected-natural-gradient-based","title":"Eigenvalue-corrected Natural Gradient Based on a New Approximation","date":"2020-11-27","arxiv_id":"2011.13609","n_code_links":0,"syntology":null},{"paper":null,"slug":"adam-a-stochastic-method-with-adaptive-1","title":"Adam$^+$: A Stochastic Method with Adaptive Variance Reduction","date":"2020-11-24","arxiv_id":"2011.11985","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-analysis-of-homotopy-sgd-for-non-1","title":"Convergence Analysis of Homotopy-SGD for non-convex optimization","date":"2020-11-20","arxiv_id":"2011.10298","n_code_links":0,"syntology":null},{"paper":null,"slug":"gradient-regularisation-as-approximate","title":"Variational Laplace for Bayesian neural networks","date":"2020-11-20","arxiv_id":"2011.10443","n_code_links":0,"syntology":null},{"paper":"/paper/strong-data-augmentation-sanitizes-poisoning","slug":"strong-data-augmentation-sanitizes-poisoning","title":"Strong Data Augmentation Sanitizes Poisoning and Backdoor Attacks Without an Accuracy Tradeoff","date":"2020-11-18","arxiv_id":"2011.09527","n_code_links":1,"syntology":null},{"paper":null,"slug":"contrastive-weight-regularization-for-large","title":"Contrastive Weight Regularization for Large Minibatch SGD","date":"2020-11-17","arxiv_id":"2011.08968","n_code_links":0,"syntology":null},{"paper":null,"slug":"avoiding-communication-in-logistic-regression","title":"Avoiding Communication in Logistic Regression","date":"2020-11-16","arxiv_id":"2011.08281","n_code_links":0,"syntology":null},{"paper":"/paper/mixing-adam-and-sgd-a-combined-optimization","slug":"mixing-adam-and-sgd-a-combined-optimization","title":"Mixing ADAM and SGD: a Combined Optimization Method","date":"2020-11-16","arxiv_id":"2011.08042","n_code_links":1,"syntology":null},{"paper":null,"slug":"ridge-rider-finding-diverse-solutions-by-1","title":"Ridge Rider: Finding Diverse Solutions by Following Eigenvectors of the Hessian","date":"2020-11-12","arxiv_id":"2011.06505","n_code_links":0,"syntology":null},{"paper":null,"slug":"direction-matters-on-the-implicit-1","title":"Direction Matters: On the Implicit Bias of Stochastic Gradient Descent with Moderate Learning Rate","date":"2020-11-04","arxiv_id":"2011.02538","n_code_links":0,"syntology":null},{"paper":null,"slug":"gradient-based-empirical-risk-minimization","title":"Gradient-Based Empirical Risk Minimization using Local Polynomial Regression","date":"2020-11-04","arxiv_id":"2011.02522","n_code_links":0,"syntology":null},{"paper":null,"slug":"local-sgd-unified-theory-and-new-efficient","title":"Local SGD: Unified Theory and New Efficient Methods","date":"2020-11-03","arxiv_id":"2011.02828","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgb-stochastic-gradient-bound-method-for","title":"SGB: Stochastic Gradient Bound Method for Optimizing Partition Functions","date":"2020-11-03","arxiv_id":"2011.01474","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-greedy-bit-flip-training-algorithm-for","title":"A Greedy Bit-flip Training Algorithm for Binarized Knowledge Graph Embeddings","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"hogwild-over-distributed-local-data-sets-with","title":"Hogwild! over Distributed Local Data Sets with Linearly Increasing Mini-Batch Sizes","date":"2020-10-27","arxiv_id":"2010.14763","n_code_links":0,"syntology":null},{"paper":"/paper/optimal-client-sampling-for-federated","slug":"optimal-client-sampling-for-federated","title":"Optimal Client Sampling for Federated Learning","date":"2020-10-26","arxiv_id":"2010.13723","n_code_links":1,"syntology":null},{"paper":null,"slug":"local-sgd-for-saddle-point-problems","title":"Distributed Saddle-Point Problems: Lower Bounds, Near-Optimal and Robust Algorithms","date":"2020-10-25","arxiv_id":"2010.13112","n_code_links":0,"syntology":null},{"paper":"/paper/inductive-bias-of-gradient-descent-for-1","slug":"inductive-bias-of-gradient-descent-for-1","title":"Inductive Bias of Gradient Descent for Weight Normalized Smooth Homogeneous Neural Nets","date":"2020-10-24","arxiv_id":"2010.12909","n_code_links":1,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-meets","title":"Stochastic Gradient Descent Meets Distribution Regression","date":"2020-10-24","arxiv_id":"2010.12842","n_code_links":0,"syntology":null},{"paper":"/paper/adaptive-gradient-quantization-for-data","slug":"adaptive-gradient-quantization-for-data","title":"Adaptive Gradient Quantization for Data-Parallel SGD","date":"2020-10-23","arxiv_id":"2010.12460","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tabrizian/learning-to-quantize"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/linearly-converging-error-compensated-sgd","slug":"linearly-converging-error-compensated-sgd","title":"Linearly Converging Error Compensated SGD","date":"2020-10-23","arxiv_id":"2010.12292","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":3,"phrase":"0 ran · 3 unverified","official":{"repos":["eduardgorbunov/ef_sigma_k"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"paper":null,"slug":"computationally-and-statistically-efficient","title":"Computationally and Statistically Efficient Truncated Regression","date":"2020-10-22","arxiv_id":"2010.12000","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-gradient-method-with-resilience-and-1","title":"Adaptive Gradient Method with Resilience and Momentum","date":"2020-10-21","arxiv_id":"2010.11041","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-augmentation-as-stochastic-optimization-1","title":"How Data Augmentation affects Optimization for Linear Regression","date":"2020-10-21","arxiv_id":"2010.11171","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-flatter-loss-for-bias-mitigation-in-cross","title":"A Flatter Loss for Bias Mitigation in Cross-dataset Facial Age Estimation","date":"2020-10-20","arxiv_id":"2010.10368","n_code_links":0,"syntology":null},{"paper":null,"slug":"dual-averaging-is-surprisingly-effective-for-1","title":"Dual Averaging is Surprisingly Effective for Deep Learning Optimization","date":"2020-10-20","arxiv_id":"2010.10502","n_code_links":0,"syntology":null},{"paper":null,"slug":"why-are-convolutional-nets-more-sample-1","title":"Why Are Convolutional Nets More Sample-Efficient than Fully-Connected Nets?","date":"2020-10-16","arxiv_id":"2010.08515","n_code_links":0,"syntology":null},{"paper":"/paper/adabelief-optimizer-adapting-stepsizes-by-the","slug":"adabelief-optimizer-adapting-stepsizes-by-the","title":"AdaBelief Optimizer: Adapting Stepsizes by the Belief in Observed Gradients","date":"2020-10-15","arxiv_id":"2010.07468","n_code_links":8,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["juntang-zhuang/Adabelief-Optimizer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/rnn-training-along-locally-optimal","slug":"rnn-training-along-locally-optimal","title":"RNN Training along Locally Optimal Trajectories via Frank-Wolfe Algorithm","date":"2020-10-12","arxiv_id":"2010.05397","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-theoretically-understanding-why-sgd","title":"Towards Theoretically Understanding Why SGD Generalizes Better Than ADAM in Deep Learning","date":"2020-10-12","arxiv_id":"2010.05627","n_code_links":0,"syntology":null},{"paper":null,"slug":"taxonn-a-light-weight-accelerator-for-deep","title":"TaxoNN: A Light-Weight Accelerator for Deep Neural Network Training","date":"2020-10-11","arxiv_id":"2010.05197","n_code_links":0,"syntology":null},{"paper":"/paper/aegd-adaptive-gradient-decent-with-energy","slug":"aegd-adaptive-gradient-decent-with-energy","title":"AEGD: Adaptive Gradient Descent with Energy","date":"2020-10-10","arxiv_id":"2010.05109","n_code_links":1,"syntology":null},{"paper":null,"slug":"double-forward-propagation-for-memorized","title":"Double Forward Propagation for Memorized Batch Normalization","date":"2020-10-10","arxiv_id":"2010.04947","n_code_links":0,"syntology":null},{"paper":"/paper/theedhum-nandrum-dravidian-codemix-fire2020","slug":"theedhum-nandrum-dravidian-codemix-fire2020","title":"Theedhum Nandrum@Dravidian-CodeMix-FIRE2020: A Sentiment Polarity Classifier for YouTube Comments with Code-switching between Tamil, Malayalam and English","date":"2020-10-07","arxiv_id":"2010.03189","n_code_links":1,"syntology":null},{"paper":"/paper/practical-precoding-via-asynchronous","slug":"practical-precoding-via-asynchronous","title":"Practical Precoding via Asynchronous Stochastic Successive Convex Approximation","date":"2020-10-03","arxiv_id":"2010.01360","n_code_links":1,"syntology":null},{"paper":null,"slug":"variance-reduced-methods-for-machine-learning","title":"Variance-Reduced Methods for Machine Learning","date":"2020-10-02","arxiv_id":"2010.00892","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-self-supervised-learning-with","slug":"understanding-self-supervised-learning-with","title":"Understanding Self-supervised Learning with Dual Deep Networks","date":"2020-10-01","arxiv_id":"2010.00578","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["facebookresearch/luckmatters"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/understanding-the-role-of-momentum-in-non","slug":"understanding-the-role-of-momentum-in-non","title":"Momentum via Primal Averaging: Theoretical Insights and Learning Rate Schedules for Non-Convex Optimization","date":"2020-10-01","arxiv_id":"2010.00406","n_code_links":1,"syntology":null},{"paper":null,"slug":"learned-fine-tuner-for-incongruous-few-shot","title":"Learning to Generate Image Source-Agnostic Universal Adversarial Perturbations","date":"2020-09-29","arxiv_id":"2009.13714","n_code_links":0,"syntology":null},{"paper":"/paper/apollo-an-adaptive-parameter-wise-diagonal","slug":"apollo-an-adaptive-parameter-wise-diagonal","title":"Apollo: An Adaptive Parameter-wise Diagonal Quasi-Newton Method for Nonconvex Stochastic Optimization","date":"2020-09-28","arxiv_id":"2009.13586","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["XuezheMax/apollo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improved-generalization-by-noise-enhancement","title":"Improved generalization by noise enhancement","date":"2020-09-28","arxiv_id":"2009.13094","n_code_links":0,"syntology":null},{"paper":"/paper/long-tailed-classification-by-keeping-the-1","slug":"long-tailed-classification-by-keeping-the-1","title":"Long-Tailed Classification by Keeping the Good and Removing the Bad Momentum Causal Effect","date":"2020-09-28","arxiv_id":"2009.12991","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["KaihuaTang/Long-Tailed-Recognition.pytorch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"on-efficient-constructions-of-checkpoints-1","title":"On Efficient Constructions of Checkpoints","date":"2020-09-28","arxiv_id":"2009.13003","n_code_links":0,"syntology":null},{"paper":"/paper/over-the-air-federated-learning-from","slug":"over-the-air-federated-learning-from","title":"Over-the-Air Federated Learning from Heterogeneous Data","date":"2020-09-27","arxiv_id":"2009.12787","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-many-factors-influence-minima-in-sgd","title":"How Many Factors Influence Minima in SGD?","date":"2020-09-24","arxiv_id":"2009.11858","n_code_links":0,"syntology":null},{"paper":"/paper/anomalous-diffusion-dynamics-of-learning-in","slug":"anomalous-diffusion-dynamics-of-learning-in","title":"Anomalous diffusion dynamics of learning in deep neural networks","date":"2020-09-22","arxiv_id":"2009.10588","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ifgovh/Anomalous-diffusion-dynamics-of-SGD"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"asynchronous-distributed-optimization-with","title":"Asynchronous Distributed Optimization with Stochastic Delays","date":"2020-09-22","arxiv_id":"2009.10717","n_code_links":0,"syntology":null},{"paper":null,"slug":"sparse-communication-for-training-deep","title":"Sparse Communication for Training Deep Networks","date":"2020-09-19","arxiv_id":"2009.09271","n_code_links":0,"syntology":null},{"paper":null,"slug":"low-rank-training-of-deep-neural-networks-for","title":"Low-Rank Training of Deep Neural Networks for Emerging Memory Technology","date":"2020-09-08","arxiv_id":"2009.03887","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-analysis-of-alternating-direction-method","title":"An Analysis of Alternating Direction Method of Multipliers for Feed-forward Neural Networks","date":"2020-09-06","arxiv_id":"2009.02825","n_code_links":0,"syntology":null},{"paper":"/paper/hlsgd-hierarchical-local-sgd-with-stale","slug":"hlsgd-hierarchical-local-sgd-with-stale","title":"HPSGD: Hierarchical Parallel SGD With Stale Gradients Featuring","date":"2020-09-06","arxiv_id":"2009.02701","n_code_links":1,"syntology":null},{"paper":null,"slug":"s-sgd-symmetrical-stochastic-gradient-descent","title":"S-SGD: Symmetrical Stochastic Gradient Descent with Weight Noise Injection for Reaching Flat Minima","date":"2020-09-05","arxiv_id":"2009.02479","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-communication-compression-for-distributed","title":"On Communication Compression for Distributed Optimization on Heterogeneous Data","date":"2020-09-04","arxiv_id":"2009.02388","n_code_links":0,"syntology":null},{"paper":null,"slug":"private-weighted-random-walk-stochastic","title":"Private Weighted Random Walk Stochastic Gradient Descent","date":"2020-09-03","arxiv_id":"2009.01790","n_code_links":0,"syntology":null},{"paper":"/paper/extreme-memorization-via-scale-of","slug":"extreme-memorization-via-scale-of","title":"Extreme Memorization via Scale of Initialization","date":"2020-08-31","arxiv_id":"2008.13363","n_code_links":1,"syntology":null},{"paper":null,"slug":"root-sgd-sharp-nonasymptotics-and-asymptotic","title":"ROOT-SGD: Sharp Nonasymptotics and Near-Optimal Asymptotics in a Single Algorithm","date":"2020-08-28","arxiv_id":"2008.12690","n_code_links":0,"syntology":null},{"paper":"/paper/a-fast-and-robust-bert-based-dialogue-state","slug":"a-fast-and-robust-bert-based-dialogue-state","title":"A Fast and Robust BERT-based Dialogue State Tracker for Schema-Guided Dialogue Dataset","date":"2020-08-27","arxiv_id":"2008.12335","n_code_links":1,"syntology":null},{"paper":null,"slug":"adversarially-robust-learning-via-entropic","title":"Adversarially Robust Learning via Entropic Regularization","date":"2020-08-27","arxiv_id":"2008.12338","n_code_links":0,"syntology":null},{"paper":null,"slug":"apmsqueeze-a-communication-efficient-adam","title":"APMSqueeze: A Communication Efficient Adam-Preconditioned Momentum SGD Algorithm","date":"2020-08-26","arxiv_id":"2008.11343","n_code_links":0,"syntology":null},{"paper":null,"slug":"page-a-simple-and-optimal-probabilistic","title":"PAGE: A Simple and Optimal Probabilistic Gradient Estimator for Nonconvex Optimization","date":"2020-08-25","arxiv_id":"2008.10898","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-serverless-learning","title":"Adaptive Serverless Learning","date":"2020-08-24","arxiv_id":"2008.10422","n_code_links":0,"syntology":null},{"paper":null,"slug":"periodic-stochastic-gradient-descent-with","title":"Periodic Stochastic Gradient Descent with Momentum for Decentralized Training","date":"2020-08-24","arxiv_id":"2008.10435","n_code_links":0,"syntology":null},{"paper":"/paper/nas-bench-301-and-the-case-for-surrogate","slug":"nas-bench-301-and-the-case-for-surrogate","title":"Surrogate NAS Benchmarks: Going Beyond the Limited Search Spaces of Tabular NAS Benchmarks","date":"2020-08-22","arxiv_id":"2008.09777","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["automl/nasbench301"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-dp-2-sgd-asynchronous-decentralized","title":"A(DP)$^2$SGD: Asynchronous Decentralized Parallel Stochastic Gradient Descent with Differential Privacy","date":"2020-08-21","arxiv_id":"2008.09246","n_code_links":0,"syntology":null},{"paper":"/paper/optimization-of-graph-neural-networks-with","slug":"optimization-of-graph-neural-networks-with","title":"Optimization of Graph Neural Networks with Natural Gradient Descent","date":"2020-08-21","arxiv_id":"2008.09624","n_code_links":1,"syntology":null},{"paper":"/paper/obtaining-adjustable-regularization-for-free","slug":"obtaining-adjustable-regularization-for-free","title":"Obtaining Adjustable Regularization for Free via Iterate Averaging","date":"2020-08-15","arxiv_id":"2008.06736","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["uuujf/IterAvg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"orthogonalized-sgd-and-nested-architectures","title":"Orthogonalized SGD and Nested Architectures for Anytime Neural Networks","date":"2020-08-15","arxiv_id":"2008.06635","n_code_links":0,"syntology":null},{"paper":null,"slug":"dimension-independence-in-unconstrained","title":"Fast Dimension Independent Private AdaGrad on Publicly Estimated Subspaces","date":"2020-08-14","arxiv_id":"2008.06570","n_code_links":0,"syntology":null},{"paper":null,"slug":"privacy-preserving-asynchronous-federated","title":"Privacy-Preserving Asynchronous Federated Learning Algorithms for Multi-Party Vertically Collaborative Learning","date":"2020-08-14","arxiv_id":"2008.06233","n_code_links":0,"syntology":null},{"paper":null,"slug":"three-variants-of-differential-privacy","title":"Three Variants of Differential Privacy: Lossless Conversion and Applications","date":"2020-08-14","arxiv_id":"2008.06529","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-networks-with-fast-retraining","title":"Deep Networks with Fast Retraining","date":"2020-08-13","arxiv_id":"2008.07387","n_code_links":0,"syntology":null},{"paper":null,"slug":"holdout-sgd-byzantine-tolerant-federated","title":"Holdout SGD: Byzantine Tolerant Federated Learning","date":"2020-08-11","arxiv_id":"2008.04612","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-improved-convergence-analysis-for","title":"An improved convergence analysis for decentralized online stochastic non-convex optimization","date":"2020-08-10","arxiv_id":"2008.04195","n_code_links":0,"syntology":null},{"paper":"/paper/mime-mimicking-centralized-stochastic","slug":"mime-mimicking-centralized-stochastic","title":"Mime: Mimicking Centralized Stochastic Algorithms in Federated Learning","date":"2020-08-08","arxiv_id":"2008.03606","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"meta-lr-schedule-net-learned-lr-schedules","title":"MLR-SNet: Transferable LR Schedules for Heterogeneous Tasks","date":"2020-07-29","arxiv_id":"2007.14546","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-high-probability-analysis-of-adaptive-sgd","title":"A High Probability Analysis of Adaptive SGD with Momentum","date":"2020-07-28","arxiv_id":"2007.14294","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-normalized-gradient-descent-with","title":"Stochastic Normalized Gradient Descent with Momentum for Large-Batch Training","date":"2020-07-28","arxiv_id":"2007.13985","n_code_links":0,"syntology":null},{"paper":"/paper/multi-level-local-sgd-for-heterogeneous","slug":"multi-level-local-sgd-for-heterogeneous","title":"Multi-Level Local SGD for Heterogeneous Hierarchical Networks","date":"2020-07-27","arxiv_id":"2007.13819","n_code_links":1,"syntology":{"ran":10,"of":17,"n_ran_checked":5,"n_instrument":5,"unverified":7,"pointer_only":17,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 3 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 7 unverified","official":{"repos":["rpi-nsl/MLL-SGD"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"stochastic-gradient-descent-applied-to-least","title":"On the Regularization Effect of Stochastic Gradient Descent applied to Least Squares","date":"2020-07-27","arxiv_id":"2007.13288","n_code_links":0,"syntology":null},{"paper":null,"slug":"cser-communication-efficient-sgd-with-error","title":"CSER: Communication-efficient SGD with Error Reset","date":"2020-07-26","arxiv_id":"2007.13221","n_code_links":0,"syntology":null},{"paper":"/paper/train-like-a-var-pro-efficient-training-of","slug":"train-like-a-var-pro-efficient-training-of","title":"Train Like a (Var)Pro: Efficient Training of Neural Networks with Variable Projection","date":"2020-07-26","arxiv_id":"2007.13171","n_code_links":1,"syntology":null},{"paper":"/paper/economical-ensembles-with-hypernetworks","slug":"economical-ensembles-with-hypernetworks","title":"Neural networks with late-phase weights","date":"2020-07-25","arxiv_id":"2007.12927","n_code_links":2,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["google/uncertainty-baselines","seijin-kobayashi/late-phase-weights"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"how-to-democratise-and-protect-ai-fair-and","title":"How to Democratise and Protect AI: Fair and Differentially Private Decentralised Deep Learning","date":"2020-07-18","arxiv_id":"2007.09370","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-regularization-of-gradient-descent-layer","title":"On regularization of gradient descent, layer imbalance and flat minima","date":"2020-07-18","arxiv_id":"2007.09286","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributed-reinforcement-learning-of","title":"Distributed Reinforcement Learning of Targeted Grasping with Active Vision for Mobile Manipulators","date":"2020-07-16","arxiv_id":"2007.08082","n_code_links":0,"syntology":null},{"paper":null,"slug":"analysis-of-q-learning-with-adaptation-and","title":"Analysis of Q-learning with Adaptation and Momentum Restart for Gradient Descent","date":"2020-07-15","arxiv_id":"2007.07422","n_code_links":0,"syntology":null},{"paper":"/paper/non-greedy-gradient-based-hyperparameter","slug":"non-greedy-gradient-based-hyperparameter","title":"Gradient-based Hyperparameter Optimization Over Long Horizons","date":"2020-07-15","arxiv_id":"2007.07869","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-periodic-averaging-a-practical","title":"Adaptive Periodic Averaging: A Practical Approach to Reducing Communication in Distributed Learning","date":"2020-07-13","arxiv_id":"2007.06134","n_code_links":0,"syntology":null}],"record_sha256":"35500054828d263cdce53d0ade2809ffbe4c8dc9bec576a4b2912a1e8a832793","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}