{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/18","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":18,"pages_in_order":21,"rows_per_page":100,"rows":[1701,1800],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/17","next":"/method/sgd/papers/19","papers":[{"paper":null,"slug":"a-scalable-framework-for-acceleration-of-cnn","title":"FPDeep: Scalable Acceleration of CNN Training on Deeply-Pipelined FPGA Clusters","date":"2019-01-04","arxiv_id":"1901.01007","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-converges-to-global-minimum-in-deep","title":"SGD Converges to Global Minimum in Deep Learning via Star-convex Path","date":"2019-01-02","arxiv_id":"1901.00451","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-continuous-time-analysis-of-distributed","title":"A continuous-time analysis of distributed stochastic gradient","date":"2018-12-28","arxiv_id":"1812.10995","n_code_links":0,"syntology":null},{"paper":null,"slug":"overparameterized-nonlinear-learning-gradient","title":"Overparameterized Nonlinear Learning: Gradient Descent Takes the Shortest Path?","date":"2018-12-25","arxiv_id":"1812.10004","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-doubly-robust-gradient","title":"Stochastic Doubly Robust Gradient","date":"2018-12-21","arxiv_id":"1812.08997","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-online-learning-via-meta-learning","title":"Deep Online Learning via Meta-Learning: Continual Adaptation for Model-Based RL","date":"2018-12-18","arxiv_id":"1812.07671","n_code_links":0,"syntology":null},{"paper":null,"slug":"provable-limitations-of-deep-learning","title":"Provable limitations of deep learning","date":"2018-12-16","arxiv_id":"1812.06369","n_code_links":0,"syntology":null},{"paper":null,"slug":"stagewise-training-accelerates-convergence-of","title":"Stagewise Training Accelerates Convergence of Testing Error Over SGD","date":"2018-12-10","arxiv_id":"1812.03934","n_code_links":0,"syntology":null},{"paper":"/paper/weighted-risk-minimization-deep-learning","slug":"weighted-risk-minimization-deep-learning","title":"What is the Effect of Importance Weighting in Deep Learning?","date":"2018-12-08","arxiv_id":"1812.03372","n_code_links":1,"syntology":null},{"paper":"/paper/elastic-gossip-distributing-neural-network","slug":"elastic-gossip-distributing-neural-network","title":"Elastic Gossip: Distributing Neural Network Training Using Gossip-like Protocols","date":"2018-12-06","arxiv_id":"1812.02407","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-theoretical-understanding-of-large","title":"Towards Theoretical Understanding of Large Batch Training in Stochastic Gradient Descent","date":"2018-12-03","arxiv_id":"1812.00542","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-linear-speedup-analysis-of-distributed-deep","title":"A Linear Speedup Analysis of Distributed Deep Learning with Sparse and Quantized Communication","date":"2018-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"exact-natural-gradient-in-deep-linear","title":"Exact natural gradient in deep linear networks and its application to the nonlinear case","date":"2018-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/how-sgd-selects-the-global-minima-in-over","slug":"how-sgd-selects-the-global-minima-in-over","title":"How SGD Selects the Global Minima in Over-parameterized Learning: A Dynamical Stability Perspective","date":"2018-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-local-hessian-in-back-propagation","title":"On the Local Hessian in Back-propagation","date":"2018-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-composite-mirror-descent-optimal","title":"Stochastic Composite Mirror Descent: Optimal Bounds with High Probabilities","date":"2018-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-primal-dual-method-for-empirical","title":"Stochastic Primal-Dual Method for Empirical Risk Minimization with O(1) Per-Iteration Complexity","date":"2018-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"training-deep-models-faster-with-robust","title":"Training Deep Models Faster with Robust, Approximate Importance Sampling","date":"2018-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"variance-reduced-stochastic-gradient-descent","title":"Variance-Reduced Stochastic Gradient Descent on Streaming Data","date":"2018-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/espnetv2-a-light-weight-power-efficient-and","slug":"espnetv2-a-light-weight-power-efficient-and","title":"ESPNetv2: A Light-weight, Power Efficient, and General Purpose Convolutional Neural Network","date":"2018-11-28","arxiv_id":"1811.11431","n_code_links":10,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sacmehta/EdgeNets"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/stochastic-gradient-push-for-distributed-deep","slug":"stochastic-gradient-push-for-distributed-deep","title":"Stochastic Gradient Push for Distributed Deep Learning","date":"2018-11-27","arxiv_id":"1811.10792","n_code_links":3,"syntology":{"ran":10,"of":15,"n_ran_checked":10,"n_instrument":0,"unverified":5,"pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":null,"slug":"the-promises-and-pitfalls-of-stochastic","title":"The promises and pitfalls of Stochastic Gradient Langevin Dynamics","date":"2018-11-25","arxiv_id":"1811.10072","n_code_links":0,"syntology":null},{"paper":"/paper/hydra-a-peer-to-peer-distributed-training","slug":"hydra-a-peer-to-peer-distributed-training","title":"Hydra: A Peer to Peer Distributed Training & Data Collection Framework","date":"2018-11-24","arxiv_id":"1811.09878","n_code_links":1,"syntology":null},{"paper":"/paper/hyperadam-a-learnable-task-adaptive-adam-for","slug":"hyperadam-a-learnable-task-adaptive-adam-for","title":"HyperAdam: A Learnable Task-Adaptive Adam for Network Training","date":"2018-11-22","arxiv_id":"1811.08996","n_code_links":2,"syntology":null},{"paper":"/paper/deep-frank-wolfe-for-neural-network","slug":"deep-frank-wolfe-for-neural-network","title":"Deep Frank-Wolfe For Neural Network Optimization","date":"2018-11-19","arxiv_id":"1811.07591","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["oval-group/dfw"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"minimum-weight-norm-models-do-not-always","title":"Minimum weight norm models do not always generalize well for over-parameterized problems","date":"2018-11-16","arxiv_id":"1811.07055","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-and-generalization-in","title":"Learning and Generalization in Overparameterized Neural Networks, Going Beyond Two Layers","date":"2018-11-12","arxiv_id":"1811.04918","n_code_links":0,"syntology":null},{"paper":null,"slug":"new-convergence-aspects-of-stochastic","title":"New Convergence Aspects of Stochastic Gradient Algorithms","date":"2018-11-10","arxiv_id":"1811.12403","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-convergence-theory-for-deep-learning-via","title":"A Convergence Theory for Deep Learning via Over-Parameterization","date":"2018-11-09","arxiv_id":"1811.03962","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-exponential-convergence-of-sgd-in-non","title":"On exponential convergence of SGD in non-convex over-parametrized learning","date":"2018-11-06","arxiv_id":"1811.02564","n_code_links":0,"syntology":null},{"paper":null,"slug":"quasi-newton-optimization-in-deep-q-learning","title":"Deep Reinforcement Learning via L-BFGS Optimization","date":"2018-11-06","arxiv_id":"1811.02693","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-modified-equations-and-dynamics-of","title":"Stochastic Modified Equations and Dynamics of Stochastic Gradient Algorithms I: Mathematical Foundations","date":"2018-11-05","arxiv_id":"1811.01558","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-primal-dual-method-for-empirical-1","title":"Stochastic Primal-Dual Method for Empirical Risk Minimization with $\\mathcal{O}(1)$ Per-Iteration Complexity","date":"2018-11-03","arxiv_id":"1811.01182","n_code_links":0,"syntology":null},{"paper":null,"slug":"implicit-regularization-of-stochastic","title":"Implicit Regularization of Stochastic Gradient Descent in Natural Language Processing: Observations and Implications","date":"2018-11-01","arxiv_id":"1811.00659","n_code_links":0,"syntology":null},{"paper":"/paper/accelerating-stochastic-training-for-over","slug":"accelerating-stochastic-training-for-over","title":"Accelerating SGD with momentum for over-parameterized learning","date":"2018-10-31","arxiv_id":"1810.13395","n_code_links":1,"syntology":null},{"paper":"/paper/kalman-gradient-descent-adaptive-variance","slug":"kalman-gradient-descent-adaptive-variance","title":"Kalman Gradient Descent: Adaptive Variance Reduction in Stochastic Optimization","date":"2018-10-29","arxiv_id":"1810.12273","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-convergence-rate-of-training-recurrent","title":"On the Convergence Rate of Training Recurrent Neural Networks","date":"2018-10-29","arxiv_id":"1810.12065","n_code_links":0,"syntology":null},{"paper":null,"slug":"finding-mixed-nash-equilibria-of-generative","title":"Finding Mixed Nash Equilibria of Generative Adversarial Networks","date":"2018-10-23","arxiv_id":"1811.02002","n_code_links":0,"syntology":null},{"paper":"/paper/ensmallen-a-flexible-c-library-for-efficient","slug":"ensmallen-a-flexible-c-library-for-efficient","title":"ensmallen: a flexible C++ library for efficient function optimization","date":"2018-10-22","arxiv_id":"1810.09361","n_code_links":1,"syntology":null},{"paper":null,"slug":"optimality-of-the-final-model-found-via","title":"Optimality of the final model found via Stochastic Gradient Descent","date":"2018-10-22","arxiv_id":"1810.09418","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-communication-strategies-to-achieve","title":"Adaptive Communication Strategies to Achieve the Best Error-Runtime Trade-off in Local-Update SGD","date":"2018-10-19","arxiv_id":"1810.08313","n_code_links":0,"syntology":null},{"paper":null,"slug":"exchangeability-and-kernel-invariance-in","title":"Exchangeability and Kernel Invariance in Trained MLPs","date":"2018-10-19","arxiv_id":"1810.08351","n_code_links":0,"syntology":null},{"paper":null,"slug":"finite-sample-expressive-power-of-small-width","title":"Small ReLU networks are powerful memorizers: a tight analysis of memorization capacity","date":"2018-10-17","arxiv_id":"1810.07770","n_code_links":0,"syntology":null},{"paper":"/paper/evolutionary-stochastic-gradient-descent-for","slug":"evolutionary-stochastic-gradient-descent-for","title":"Evolutionary Stochastic Gradient Descent for Optimization of Deep Neural Networks","date":"2018-10-16","arxiv_id":"1810.06773","n_code_links":1,"syntology":null},{"paper":null,"slug":"fast-and-faster-convergence-of-sgd-for-over","title":"Fast and Faster Convergence of SGD for Over-Parameterized Models and an Accelerated Perceptron","date":"2018-10-16","arxiv_id":"1810.07288","n_code_links":0,"syntology":null},{"paper":"/paper/quasi-hyperbolic-momentum-and-adam-for-deep","slug":"quasi-hyperbolic-momentum-and-adam-for-deep","title":"Quasi-hyperbolic momentum and Adam for deep learning","date":"2018-10-16","arxiv_id":"1810.06801","n_code_links":2,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["facebookresearch/qhoptim"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"training-deep-neural-network-in-limited","title":"Training Deep Neural Network in Limited Precision","date":"2018-10-12","arxiv_id":"1810.05486","n_code_links":0,"syntology":null},{"paper":null,"slug":"bayesian-deep-convolutional-networks-with","title":"Bayesian Deep Convolutional Networks with Many Channels are Gaussian Processes","date":"2018-10-11","arxiv_id":"1810.05148","n_code_links":0,"syntology":null},{"paper":"/paper/signsgd-with-majority-vote-is-communication","slug":"signsgd-with-majority-vote-is-communication","title":"signSGD with Majority Vote is Communication Efficient And Fault Tolerant","date":"2018-10-11","arxiv_id":"1810.05291","n_code_links":4,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"tight-dimension-independent-lower-bound-on","title":"Tight Dimension Independent Lower Bound on the Expected Convergence Rate for Diminishing Step Sizes in SGD","date":"2018-10-10","arxiv_id":"1810.04723","n_code_links":0,"syntology":null},{"paper":null,"slug":"anytime-stochastic-gradient-descent-a-time-to","title":"Anytime Stochastic Gradient Descent: A Time to Hear from all the Workers","date":"2018-10-06","arxiv_id":"1810.02976","n_code_links":0,"syntology":null},{"paper":"/paper/continuous-time-models-for-stochastic","slug":"continuous-time-models-for-stochastic","title":"Continuous-time Models for Stochastic Optimization Algorithms","date":"2018-10-05","arxiv_id":"1810.02565","n_code_links":1,"syntology":null},{"paper":"/paper/large-batch-size-training-of-neural-networks","slug":"large-batch-size-training-of-neural-networks","title":"Large batch size training of neural networks with adversarial training and second-order information","date":"2018-10-02","arxiv_id":"1810.01021","n_code_links":1,"syntology":null},{"paper":null,"slug":"directional-analysis-of-stochastic-gradient","title":"Directional Analysis of Stochastic Gradient Descent via von Mises-Fisher Distributions in Deep learning","date":"2018-09-29","arxiv_id":"1810.00150","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-convergence-of-sparsified-gradient","title":"The Convergence of Sparsified Gradient Methods","date":"2018-09-27","arxiv_id":"1809.10505","n_code_links":0,"syntology":null},{"paper":"/paper/preconditioner-on-matrix-lie-group-for-sgd","slug":"preconditioner-on-matrix-lie-group-for-sgd","title":"Preconditioner on Matrix Lie Group for SGD","date":"2018-09-26","arxiv_id":"1809.10232","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":1,"n_instrument":4,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lixilinx/psgd_torch"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/sparsified-sgd-with-memory","slug":"sparsified-sgd-with-memory","title":"Sparsified SGD with Memory","date":"2018-09-20","arxiv_id":"1809.07599","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["epfml/sparsifiedSGD"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"graph-dependent-implicit-regularisation-for","title":"Graph-Dependent Implicit Regularisation for Distributed Stochastic Subgradient Descent","date":"2018-09-18","arxiv_id":"1809.06958","n_code_links":0,"syntology":null},{"paper":null,"slug":"discovering-low-precision-networks-close-to","title":"Discovering Low-Precision Networks Close to Full-Precision Networks for Efficient Embedded Inference","date":"2018-09-11","arxiv_id":"1809.04191","n_code_links":0,"syntology":null},{"paper":"/paper/learning-rate-adaptation-for-federated-and","slug":"learning-rate-adaptation-for-federated-and","title":"Learning Rate Adaptation for Federated and Differentially Private Learning","date":"2018-09-11","arxiv_id":"1809.03832","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"privacy-preserving-deep-learning-via-weight","title":"Privacy-Preserving Deep Learning via Weight Transmission","date":"2018-09-10","arxiv_id":"1809.03272","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-learns-state","title":"Stochastic Gradient Descent Learns State Equations with Nonlinear Activations","date":"2018-09-09","arxiv_id":"1809.03019","n_code_links":0,"syntology":null},{"paper":null,"slug":"decentralized-differentially-private-without","title":"Decentralized Differentially Private Without-Replacement Stochastic Gradient Descent","date":"2018-09-08","arxiv_id":"1809.02727","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-ica-understanding-global-dynamics-of","title":"Online ICA: Understanding Global Dynamics of Nonconvex Optimization via Diffusion Processes","date":"2018-08-29","arxiv_id":"1808.09642","n_code_links":0,"syntology":null},{"paper":null,"slug":"cooperative-sgd-a-unified-framework-for-the","title":"Cooperative SGD: A unified Framework for the Design and Analysis of Communication-Efficient SGD Algorithms","date":"2018-08-22","arxiv_id":"1808.07576","n_code_links":0,"syntology":null},{"paper":"/paper/dont-use-large-mini-batches-use-local-sgd","slug":"dont-use-large-mini-batches-use-local-sgd","title":"Don't Use Large Mini-Batches, Use Local SGD","date":"2018-08-22","arxiv_id":"1808.07217","n_code_links":2,"syntology":null},{"paper":null,"slug":"universal-stagewise-learning-for-non-convex","title":"Universal Stagewise Learning for Non-Convex Problems with Convergence on Averaged Solutions","date":"2018-08-20","arxiv_id":"1808.06296","n_code_links":0,"syntology":null},{"paper":null,"slug":"ensemble-kalman-inversion-a-derivative-free","title":"Ensemble Kalman Inversion: A Derivative-Free Technique For Machine Learning Tasks","date":"2018-08-10","arxiv_id":"1808.03620","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-weighted-adagrad-with","title":"A Unified Analysis of AdaGrad with Weighted Aggregation and Momentum Acceleration","date":"2018-08-10","arxiv_id":"1808.03408","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-overparameterized-neural-networks","title":"Learning Overparameterized Neural Networks via Stochastic Gradient Descent on Structured Data","date":"2018-08-03","arxiv_id":"1808.01204","n_code_links":0,"syntology":null},{"paper":"/paper/stochastic-gradient-descent-with-biased-but","slug":"stochastic-gradient-descent-with-biased-but","title":"Stochastic Gradient Descent with Biased but Consistent Gradient Estimators","date":"2018-07-31","arxiv_id":"1807.11880","n_code_links":1,"syntology":null},{"paper":"/paper/a-surprising-linear-relationship-predicts","slug":"a-surprising-linear-relationship-predicts","title":"A Surprising Linear Relationship Predicts Test Performance in Deep Networks","date":"2018-07-25","arxiv_id":"1807.09659","n_code_links":3,"syntology":null},{"paper":null,"slug":"signprox-one-bit-proximal-algorithm-for","title":"signProx: One-Bit Proximal Algorithm for Nonconvex Stochastic Optimization","date":"2018-07-20","arxiv_id":"1807.08023","n_code_links":0,"syntology":null},{"paper":"/paper/a-unified-theory-of-adaptive-stochastic","slug":"a-unified-theory-of-adaptive-stochastic","title":"Bayesian filtering unifies adaptive and non-adaptive neural network optimization methods","date":"2018-07-19","arxiv_id":"1807.07540","n_code_links":1,"syntology":null},{"paper":null,"slug":"parallel-restarted-sgd-with-faster","title":"Parallel Restarted SGD with Faster Convergence and Less Communication: Demystifying Why Model Averaging Works for Deep Learning","date":"2018-07-17","arxiv_id":"1807.06629","n_code_links":0,"syntology":null},{"paper":null,"slug":"evolving-differentiable-gene-regulatory","title":"Evolving Differentiable Gene Regulatory Networks","date":"2018-07-16","arxiv_id":"1807.05948","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-relation-between-the-sharpest","slug":"on-the-relation-between-the-sharpest","title":"On the Relation Between the Sharpest Directions of DNN Loss and the SGD Step Length","date":"2018-07-13","arxiv_id":"1807.05031","n_code_links":1,"syntology":null},{"paper":"/paper/maximizing-invariant-data-perturbation-with","slug":"maximizing-invariant-data-perturbation-with","title":"Maximizing Invariant Data Perturbation with Stochastic Optimization","date":"2018-07-12","arxiv_id":"1807.05077","n_code_links":1,"syntology":null},{"paper":null,"slug":"metalearning-with-hebbian-fast-weights","title":"Metalearning with Hebbian Fast Weights","date":"2018-07-12","arxiv_id":"1807.05076","n_code_links":0,"syntology":null},{"paper":null,"slug":"quasi-monte-carlo-variational-inference","title":"Quasi-Monte Carlo Variational Inference","date":"2018-07-04","arxiv_id":"1807.01604","n_code_links":0,"syntology":null},{"paper":null,"slug":"batch-is-not-heavy-learning-word","title":"Batch IS NOT Heavy: Learning Word Representations From All Samples","date":"2018-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"random-shuffling-beats-sgd-after-finite","title":"Random Shuffling Beats SGD after Finite Epochs","date":"2018-06-26","arxiv_id":"1806.10077","n_code_links":0,"syntology":null},{"paper":null,"slug":"faster-sgd-training-by-minibatch-persistency","title":"Faster SGD training by minibatch persistency","date":"2018-06-19","arxiv_id":"1806.07353","n_code_links":0,"syntology":null},{"paper":"/paper/closing-the-generalization-gap-of-adaptive","slug":"closing-the-generalization-gap-of-adaptive","title":"Closing the Generalization Gap of Adaptive Gradient Methods in Training Deep Neural Networks","date":"2018-06-18","arxiv_id":"1806.06763","n_code_links":2,"syntology":null},{"paper":null,"slug":"using-mode-connectivity-for-loss-landscape","title":"Using Mode Connectivity for Loss Landscape Analysis","date":"2018-06-18","arxiv_id":"1806.06977","n_code_links":0,"syntology":null},{"paper":"/paper/there-are-many-consistent-explanations-of","slug":"there-are-many-consistent-explanations-of","title":"There Are Many Consistent Explanations of Unlabeled Data: Why You Should Average","date":"2018-06-14","arxiv_id":"1806.05594","n_code_links":2,"syntology":null},{"paper":null,"slug":"boosted-training-of-convolutional-neural","title":"Boosted Training of Convolutional Neural Networks for Multi-Class Segmentation","date":"2018-06-13","arxiv_id":"1806.05974","n_code_links":0,"syntology":null},{"paper":"/paper/when-will-gradient-methods-converge-to-max","slug":"when-will-gradient-methods-converge-to-max","title":"When Will Gradient Methods Converge to Max-margin Classifier under ReLU Models?","date":"2018-06-12","arxiv_id":"1806.04339","n_code_links":1,"syntology":null},{"paper":"/paper/dropback-continuous-pruning-during-training","slug":"dropback-continuous-pruning-during-training","title":"Full deep neural network training on a pruned weight budget","date":"2018-06-11","arxiv_id":"1806.06949","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-effect-of-network-width-on-the","title":"The Effect of Network Width on the Performance of Large-batch Training","date":"2018-06-11","arxiv_id":"1806.03791","n_code_links":0,"syntology":null},{"paper":null,"slug":"lightweight-stochastic-optimization-for","title":"Lightweight Stochastic Optimization for Minimizing Finite Sums with Infinite Data","date":"2018-06-08","arxiv_id":"1806.02927","n_code_links":0,"syntology":null},{"paper":null,"slug":"probabilistic-deep-learning-using-random-sum","title":"Probabilistic Deep Learning using Random Sum-Product Networks","date":"2018-06-05","arxiv_id":"1806.01910","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-on-separable-data","title":"Stochastic Gradient Descent on Separable Data: Exact Convergence with a Fixed Learning Rate","date":"2018-06-05","arxiv_id":"1806.01796","n_code_links":0,"syntology":null},{"paper":"/paper/backdrop-stochastic-backpropagation","slug":"backdrop-stochastic-backpropagation","title":"Backdrop: Stochastic Backpropagation","date":"2018-06-04","arxiv_id":"1806.01337","n_code_links":1,"syntology":null},{"paper":null,"slug":"geometry-aware-constrained-optimization","title":"Geometry Aware Constrained Optimization Techniques for Deep Learning","date":"2018-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"on-consensus-optimality-trade-offs-in","title":"On Consensus-Optimality Trade-offs in Collaborative Deep Learning","date":"2018-05-30","arxiv_id":"1805.12120","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-much-restricted-isometry-is-needed-in","title":"How Much Restricted Isometry is Needed In Nonconvex Matrix Recovery?","date":"2018-05-25","arxiv_id":"1805.10251","n_code_links":0,"syntology":null},{"paper":null,"slug":"statistical-optimality-of-stochastic-gradient","title":"Statistical Optimality of Stochastic Gradient Descent on Hard Learning Problems through Multiple Passes","date":"2018-05-25","arxiv_id":"1805.10074","n_code_links":0,"syntology":null},{"paper":"/paper/zeno-byzantine-suspicious-stochastic-gradient","slug":"zeno-byzantine-suspicious-stochastic-gradient","title":"Zeno: Distributed Stochastic Gradient Descent with Suspicion-based Fault-tolerance","date":"2018-05-25","arxiv_id":"1805.10032","n_code_links":1,"syntology":null},{"paper":"/paper/local-sgd-converges-fast-and-communicates","slug":"local-sgd-converges-fast-and-communicates","title":"Local SGD Converges Fast and Communicates Little","date":"2018-05-24","arxiv_id":"1805.09767","n_code_links":2,"syntology":null}],"record_sha256":"e0ec9ca0d6f2161a13618d600fa4e8fdb44e48a3ca7f3bbe9b78d374cfdc2506","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}