{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/19","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":19,"pages_in_order":21,"rows_per_page":100,"rows":[1801,1900],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/18","next":"/method/sgd/papers/20","papers":[{"paper":null,"slug":"predictive-local-smoothness-for-stochastic","title":"Predictive Local Smoothness for Stochastic Gradient Methods","date":"2018-05-23","arxiv_id":"1805.09386","n_code_links":0,"syntology":null},{"paper":"/paper/gradient-energy-matching-for-distributed","slug":"gradient-energy-matching-for-distributed","title":"Gradient Energy Matching for Distributed Asynchronous Gradient Descent","date":"2018-05-22","arxiv_id":"1805.08469","n_code_links":2,"syntology":null},{"paper":"/paper/small-steps-and-giant-leaps-minimal-newton","slug":"small-steps-and-giant-leaps-minimal-newton","title":"Small steps and giant leaps: Minimal Newton solvers for Deep Learning","date":"2018-05-21","arxiv_id":"1805.08095","n_code_links":6,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jotaf98/curveball"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/smoothout-smoothing-out-sharp-minima-to","slug":"smoothout-smoothing-out-sharp-minima-to","title":"SmoothOut: Smoothing Out Sharp Minima to Improve Generalization in Deep Learning","date":"2018-05-21","arxiv_id":"1805.07898","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["wenwei202/smoothout"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"interpolatron-interpolation-or-extrapolation","title":"Interpolatron: Interpolation or Extrapolation Schemes to Accelerate Optimization for Deep Neural Networks","date":"2018-05-17","arxiv_id":"1805.06753","n_code_links":0,"syntology":null},{"paper":null,"slug":"differential-equations-for-modeling","title":"Differential Equations for Modeling Asynchronous Algorithms","date":"2018-05-08","arxiv_id":"1805.02991","n_code_links":0,"syntology":null},{"paper":"/paper/implementation-of-stochastic-quasi-newtons","slug":"implementation-of-stochastic-quasi-newtons","title":"Implementation of Stochastic Quasi-Newton's Method in PyTorch","date":"2018-05-07","arxiv_id":"1805.02338","n_code_links":1,"syntology":null},{"paper":"/paper/a-scalable-discrete-time-survival-model-for","slug":"a-scalable-discrete-time-survival-model-for","title":"A Scalable Discrete-Time Survival Model for Neural Networks","date":"2018-05-02","arxiv_id":"1805.00917","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["MGensheimer/nnet-survival"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"k-svrg-variance-reduction-for-large-scale","title":"k-SVRG: Variance Reduction for Large Scale Optimization","date":"2018-05-02","arxiv_id":"1805.00982","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-networks-as-interacting-particle","title":"Trainability and Accuracy of Neural Networks: An Interacting Particle System Approach","date":"2018-05-02","arxiv_id":"1805.00915","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-representation-ensembles-and-delayed","title":"Multi-representation Ensembles and Delayed SGD Updates Improve Syntax-based NMT","date":"2018-05-01","arxiv_id":"1805.00456","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-mean-field-view-of-the-landscape-of-two","title":"A Mean Field View of the Landscape of Two-Layers Neural Networks","date":"2018-04-18","arxiv_id":"1804.06561","n_code_links":0,"syntology":null},{"paper":"/paper/active-mini-batch-sampling-using-repulsive","slug":"active-mini-batch-sampling-using-repulsive","title":"Active Mini-Batch Sampling using Repulsive Point Processes","date":"2018-04-08","arxiv_id":"1804.02772","n_code_links":1,"syntology":null},{"paper":null,"slug":"byzantine-stochastic-gradient-descent","title":"Byzantine Stochastic Gradient Descent","date":"2018-03-23","arxiv_id":"1803.08917","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-convergence-of-stochastic-gradient","title":"The Convergence of Stochastic Gradient Descent in Asynchronous Shared Memory","date":"2018-03-23","arxiv_id":"1803.08841","n_code_links":0,"syntology":null},{"paper":null,"slug":"lower-error-bounds-for-the-stochastic","title":"Lower error bounds for the stochastic gradient descent optimization algorithm: Sharp convergence rates for slowly and fast decaying learning rates","date":"2018-03-22","arxiv_id":"1803.08600","n_code_links":0,"syntology":null},{"paper":null,"slug":"escaping-saddles-with-stochastic-gradients","title":"Escaping Saddles with Stochastic Gradients","date":"2018-03-15","arxiv_id":"1803.05999","n_code_links":0,"syntology":null},{"paper":null,"slug":"gossipgrad-scalable-deep-learning-using","title":"GossipGraD: Scalable Deep Learning using Gossip Communication based Asynchronous Gradient Descent","date":"2018-03-15","arxiv_id":"1803.05880","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-insufficiency-of-existing-momentum","slug":"on-the-insufficiency-of-existing-momentum","title":"On the insufficiency of existing momentum schemes for Stochastic Optimization","date":"2018-03-15","arxiv_id":"1803.05591","n_code_links":2,"syntology":null},{"paper":"/paper/averaging-weights-leads-to-wider-optima-and","slug":"averaging-weights-leads-to-wider-optima-and","title":"Averaging Weights Leads to Wider Optima and Better Generalization","date":"2018-03-14","arxiv_id":"1803.05407","n_code_links":17,"syntology":{"ran":7,"of":9,"n_ran_checked":5,"n_instrument":2,"unverified":2,"pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["timgaripov/swa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"lsh-microbatches-for-stochastic-gradients","title":"Self-Similar Epochs: Value in Arrangement","date":"2018-03-14","arxiv_id":"1803.05389","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-agnostic-private-learning-via-stability","title":"Model-Agnostic Private Learning via Stability","date":"2018-03-14","arxiv_id":"1803.05101","n_code_links":0,"syntology":null},{"paper":"/paper/high-accuracy-low-precision-training","slug":"high-accuracy-low-precision-training","title":"High-Accuracy Low-Precision Training","date":"2018-03-09","arxiv_id":"1803.03383","n_code_links":1,"syntology":null},{"paper":null,"slug":"fast-convergence-for-stochastic-and","title":"Fast Convergence for Stochastic and Distributed Gradient Descent in the Interpolation Limit","date":"2018-03-08","arxiv_id":"1803.02922","n_code_links":0,"syntology":null},{"paper":null,"slug":"slow-and-stale-gradients-can-win-the-race","title":"Slow and Stale Gradients Can Win the Race: Error-Runtime Trade-offs in Distributed SGD","date":"2018-03-03","arxiv_id":"1803.01113","n_code_links":0,"syntology":null},{"paper":"/paper/not-all-samples-are-created-equal-deep","slug":"not-all-samples-are-created-equal-deep","title":"Not All Samples Are Created Equal: Deep Learning with Importance Sampling","date":"2018-03-02","arxiv_id":"1803.00942","n_code_links":2,"syntology":null},{"paper":"/paper/the-anisotropic-noise-in-stochastic-gradient","slug":"the-anisotropic-noise-in-stochastic-gradient","title":"The Anisotropic Noise in Stochastic Gradient Descent: Its Behavior of Escaping from Sharp Minima and Regularization Effects","date":"2018-03-01","arxiv_id":"1803.00195","n_code_links":1,"syntology":null},{"paper":null,"slug":"train-feedfoward-neural-network-with-layer","title":"Train Feedfoward Neural Network with Layer-wise Adaptive Rate via Approximating Back-matching Propagation","date":"2018-02-27","arxiv_id":"1802.09750","n_code_links":0,"syntology":null},{"paper":"/paper/shampoo-preconditioned-stochastic-tensor","slug":"shampoo-preconditioned-stochastic-tensor","title":"Shampoo: Preconditioned Stochastic Tensor Optimization","date":"2018-02-26","arxiv_id":"1802.09568","n_code_links":3,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"a-walk-with-sgd","title":"A Walk with SGD","date":"2018-02-24","arxiv_id":"1802.08770","n_code_links":0,"syntology":null},{"paper":"/paper/stochastic-gradient-descent-on-highly","slug":"stochastic-gradient-descent-on-highly","title":"Stochastic Gradient Descent on Highly-Parallel Architectures","date":"2018-02-24","arxiv_id":"1802.08800","n_code_links":2,"syntology":null},{"paper":"/paper/asynchronous-byzantine-machine-learning-the","slug":"asynchronous-byzantine-machine-learning-the","title":"Asynchronous Byzantine Machine Learning (the case of SGD)","date":"2018-02-22","arxiv_id":"1802.07928","n_code_links":1,"syntology":null},{"paper":"/paper/the-hidden-vulnerability-of-distributed","slug":"the-hidden-vulnerability-of-distributed","title":"The Hidden Vulnerability of Distributed Learning in Byzantium","date":"2018-02-22","arxiv_id":"1802.07927","n_code_links":1,"syntology":null},{"paper":null,"slug":"generalization-error-bounds-with","title":"Generalization Error Bounds with Probabilistic Guarantee for SGD in Nonconvex Optimization","date":"2018-02-19","arxiv_id":"1802.06903","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-alternative-view-when-does-sgd-escape","title":"An Alternative View: When Does SGD Escape Local Minima?","date":"2018-02-17","arxiv_id":"1802.06175","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-role-of-information-complexity-and","title":"The Role of Information Complexity and Randomization in Representation Learning","date":"2018-02-14","arxiv_id":"1802.05355","n_code_links":0,"syntology":null},{"paper":"/paper/signsgd-compressed-optimisation-for-non","slug":"signsgd-compressed-optimisation-for-non","title":"signSGD: Compressed Optimisation for Non-Convex Problems","date":"2018-02-13","arxiv_id":"1802.04434","n_code_links":6,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jxbz/signSGD"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"uncertainty-quantification-for-online","title":"HiGrad: Uncertainty Quantification for Online Learning and Stochastic Approximation","date":"2018-02-13","arxiv_id":"1802.04876","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-and-hogwild-convergence-without-the","title":"SGD and Hogwild! Convergence Without the Bounded Gradients Assumption","date":"2018-02-11","arxiv_id":"1802.03801","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-predictor-corrector-method-for-the-training","title":"A predictor-corrector method for the training of deep neural networks","date":"2018-01-19","arxiv_id":"1803.05779","n_code_links":0,"syntology":null},{"paper":null,"slug":"when-does-stochastic-gradient-algorithm-work","title":"When Does Stochastic Gradient Algorithm Work Well?","date":"2018-01-18","arxiv_id":"1801.06159","n_code_links":0,"syntology":null},{"paper":null,"slug":"mxnet-mpi-embedding-mpi-parallelism-in","title":"MXNET-MPI: Embedding MPI parallelism in Parameter Server Task Model for scaling Deep Learning","date":"2018-01-11","arxiv_id":"1801.03855","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-to-make-the-gradients-small","title":"How To Make the Gradients Small Stochastically: Even Faster Convex and Nonconvex SGD","date":"2018-01-08","arxiv_id":"1801.02982","n_code_links":0,"syntology":null},{"paper":null,"slug":"theory-of-deep-learning-iib-optimization","title":"Theory of Deep Learning IIb: Optimization Properties of SGD","date":"2018-01-07","arxiv_id":"1801.02254","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparison-of-second-order-methods-for-deep","title":"A comparison of second-order methods for deep convolutional neural networks","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-painless-attention-mechanism-for","title":"A Painless Attention Mechanism for Convolutional Neural Networks","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"better-generalization-by-efficient-trust","title":"Better Generalization by Efficient Trust Region Method","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-rate-of-sign-stochastic-gradient","title":"Convergence rate of sign stochastic gradient descent for non-convex functions","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"demystifying-overcomplete-nonlinear-auto","title":"Demystifying overcomplete nonlinear auto-encoders: fast SGD convergence towards sparse representation from random initialization","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"faster-distributed-synchronous-sgd-with-weak","title":"Faster Distributed Synchronous SGD with Weak Synchronization","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"fixing-weight-decay-regularization-in-adam","title":"Fixing Weight Decay Regularization in Adam","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"kronecker-factored-curvature-approximations","title":"Kronecker-factored Curvature Approximations for Recurrent Neural Networks","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"lsh-sampling-breaks-the-computational-chicken","title":"LSH-SAMPLING BREAKS THE COMPUTATIONAL CHICKEN-AND-EGG LOOP IN ADAPTIVE STOCHASTIC GRADIENT ESTIMATION","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sparse-regularized-deep-neural-networks-for","title":"Sparse Regularized Deep Neural Networks For Efficient Embedded Learning","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"true-asymptotic-natural-gradient-optimization","title":"True Asymptotic Natural Gradient Optimization","date":"2017-12-22","arxiv_id":"1712.08449","n_code_links":0,"syntology":null},{"paper":"/paper/improving-generalization-performance-by","slug":"improving-generalization-performance-by","title":"Improving Generalization Performance by Switching from Adam to SGD","date":"2017-12-20","arxiv_id":"1712.07628","n_code_links":6,"syntology":null},{"paper":null,"slug":"on-the-relationship-between-the-openai","title":"On the Relationship Between the OpenAI Evolution Strategy and Stochastic Gradient Descent","date":"2017-12-18","arxiv_id":"1712.06564","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-power-of-interpolation-understanding-the","title":"The Power of Interpolation: Understanding the Effectiveness of SGD in Modern Over-parametrized Learning","date":"2017-12-18","arxiv_id":"1712.06559","n_code_links":0,"syntology":null},{"paper":"/paper/deep-gradient-compression-reducing-the","slug":"deep-gradient-compression-reducing-the","title":"Deep Gradient Compression: Reducing the Communication Bandwidth for Distributed Training","date":"2017-12-05","arxiv_id":"1712.01887","n_code_links":6,"syntology":null},{"paper":"/paper/machine-learning-with-adversaries-byzantine","slug":"machine-learning-with-adversaries-byzantine","title":"Machine Learning with Adversaries: Byzantine Tolerant Gradient Descent","date":"2017-12-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"nonlinear-acceleration-of-stochastic","title":"Nonlinear Acceleration of Stochastic Algorithms","date":"2017-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"online-to-offline-conversions-universality-1","title":"Online to Offline Conversions, Universality and Adaptive Minibatch Sizes","date":"2017-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-optimization-with-variance-1","title":"Stochastic Optimization with Variance Reduction for Infinite Datasets with Finite Sum Structure","date":"2017-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"neon2-finding-local-minima-via-first-order","title":"Neon2: Finding Local Minima via First-Order Oracles","date":"2017-11-17","arxiv_id":"1711.06673","n_code_links":0,"syntology":null},{"paper":"/paper/decoupled-weight-decay-regularization","slug":"decoupled-weight-decay-regularization","title":"Decoupled Weight Decay Regularization","date":"2017-11-14","arxiv_id":"1711.05101","n_code_links":23,"syntology":{"ran":18,"of":23,"n_ran_checked":18,"n_instrument":0,"unverified":5,"pointer_only":5,"phrase":"18 ran (of which 6 constructed an object rather than computing a result; 18 with no instrument failure: 3 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["GLambard/AdamW_Keras","Yagami123/Caffe-AdamW-AdamWR","loshchil/AdamW-and-SGDW"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"three-factors-influencing-minima-in-sgd","title":"Three Factors Influencing Minima in SGD","date":"2017-11-13","arxiv_id":"1711.04623","n_code_links":0,"syntology":null},{"paper":null,"slug":"analysis-of-approximate-stochastic-gradient","title":"Analysis of Biased Stochastic Gradient Descent Using Sequential Semidefinite Programs","date":"2017-11-03","arxiv_id":"1711.00987","n_code_links":0,"syntology":null},{"paper":"/paper/dont-decay-the-learning-rate-increase-the","slug":"dont-decay-the-learning-rate-increase-the","title":"Don't Decay the Learning Rate, Increase the Batch Size","date":"2017-11-01","arxiv_id":"1711.00489","n_code_links":3,"syntology":null},{"paper":null,"slug":"linearly-convergent-stochastic-heavy-ball","title":"Linearly convergent stochastic heavy ball method for minimizing generalization error","date":"2017-10-30","arxiv_id":"1710.10737","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-performs","title":"Stochastic gradient descent performs variational inference, converges to limit cycles for deep networks","date":"2017-10-30","arxiv_id":"1710.11029","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-learns-over-parameterized-networks-that","title":"SGD Learns Over-parameterized Networks that Provably Generalize on Linearly Separable Data","date":"2017-10-27","arxiv_id":"1710.10174","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-negative-sampling-for-word","title":"Improving Negative Sampling for Word Representation using Self-embedded Features","date":"2017-10-26","arxiv_id":"1710.09805","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-markov-chain-theory-approach-to","title":"A Markov Chain Theory Approach to Characterizing the Minimax Optimality of Stochastic Gradient Descent (for Least Squares)","date":"2017-10-25","arxiv_id":"1710.09430","n_code_links":0,"syntology":null},{"paper":null,"slug":"stability-and-generalization-of-learning","title":"Stability and Generalization of Learning Algorithms that Converge to Global Optima","date":"2017-10-23","arxiv_id":"1710.08402","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-novel-stochastic-stratified-average","title":"A Novel Stochastic Stratified Average Gradient Method: Convergence Rate and Its Complexity","date":"2017-10-21","arxiv_id":"1710.07783","n_code_links":0,"syntology":null},{"paper":"/paper/asynchronous-decentralized-parallel","slug":"asynchronous-decentralized-parallel","title":"Asynchronous Decentralized Parallel Stochastic Gradient Descent","date":"2017-10-18","arxiv_id":"1710.06952","n_code_links":3,"syntology":null},{"paper":"/paper/synkhronos-a-multi-gpu-theano-extension-for","slug":"synkhronos-a-multi-gpu-theano-extension-for","title":"Synkhronos: a Multi-GPU Theano Extension for Data Parallelism","date":"2017-10-11","arxiv_id":"1710.04162","n_code_links":1,"syntology":null},{"paper":null,"slug":"convergence-analysis-of-distributed","title":"Convergence Analysis of Distributed Stochastic Gradient Descent with Shuffling","date":"2017-09-29","arxiv_id":"1709.10432","n_code_links":0,"syntology":null},{"paper":null,"slug":"probabilistic-synchronous-parallel","title":"Probabilistic Synchronous Parallel","date":"2017-09-22","arxiv_id":"1709.07772","n_code_links":0,"syntology":null},{"paper":"/paper/neural-optimizer-search-with-reinforcement","slug":"neural-optimizer-search-with-reinforcement","title":"Neural Optimizer Search with Reinforcement Learning","date":"2017-09-21","arxiv_id":"1709.07417","n_code_links":2,"syntology":null},{"paper":"/paper/a-pac-bayesian-analysis-of-randomized","slug":"a-pac-bayesian-analysis-of-randomized","title":"A PAC-Bayesian Analysis of Randomized Learning with Application to Stochastic Gradient Descent","date":"2017-09-19","arxiv_id":"1709.06617","n_code_links":1,"syntology":null},{"paper":null,"slug":"scalable-support-vector-clustering-using","title":"Scalable Support Vector Clustering Using Budget","date":"2017-09-19","arxiv_id":"1709.06444","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-impact-of-local-geometry-and-batch-size","title":"The Impact of Local Geometry and Batch Size on Stochastic Gradient Descent for Nonconvex Problems","date":"2017-09-14","arxiv_id":"1709.04718","n_code_links":0,"syntology":null},{"paper":"/paper/normalized-direction-preserving-adam","slug":"normalized-direction-preserving-adam","title":"Normalized Direction-preserving Adam","date":"2017-09-13","arxiv_id":"1709.04546","n_code_links":1,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-going-as-fast-as","title":"Stochastic Gradient Descent: Going As Fast As Possible But Not Faster","date":"2017-09-05","arxiv_id":"1709.01427","n_code_links":0,"syntology":null},{"paper":null,"slug":"natasha-2-faster-non-convex-optimization-than","title":"Natasha 2: Faster Non-Convex Optimization Than SGD","date":"2017-08-29","arxiv_id":"1708.08694","n_code_links":0,"syntology":null},{"paper":null,"slug":"second-order-optimization-for-non-convex","title":"Second-Order Optimization for Non-Convex Machine Learning: An Empirical Study","date":"2017-08-25","arxiv_id":"1708.07827","n_code_links":0,"syntology":null},{"paper":null,"slug":"weighted-parallel-sgd-for-distributed","title":"Weighted parallel SGD for distributed unbalanced-workload training system","date":"2017-08-16","arxiv_id":"1708.04801","n_code_links":0,"syntology":null},{"paper":null,"slug":"noisy-softmax-improving-the-generalization","title":"Noisy Softmax: Improving the Generalization Ability of DCNN via Postponing the Early Softmax Saturation","date":"2017-08-12","arxiv_id":"1708.03769","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-optimization-with-bandit-sampling","title":"Stochastic Optimization with Bandit Sampling","date":"2017-08-08","arxiv_id":"1708.02544","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-optimizer-search-using-reinforcement","title":"Neural Optimizer Search using Reinforcement Learning","date":"2017-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-adaptive-quasi-newton-methods-for","title":"Stochastic Adaptive Quasi-Newton Methods for Minimizing Expected Values","date":"2017-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"bridging-the-gap-between-constant-step-size","title":"Bridging the Gap between Constant Step Size Stochastic Gradient Descent and Markov Chains","date":"2017-07-20","arxiv_id":"1707.06386","n_code_links":0,"syntology":null},{"paper":"/paper/block-normalized-gradient-method-an-empirical","slug":"block-normalized-gradient-method-an-empirical","title":"Block-Normalized Gradient Method: An Empirical Study for Training Deep Neural Network","date":"2017-07-16","arxiv_id":"1707.04822","n_code_links":2,"syntology":null},{"paper":"/paper/dual-path-networks","slug":"dual-path-networks","title":"Dual Path Networks","date":"2017-07-06","arxiv_id":"1707.01629","n_code_links":18,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"parle-parallelizing-stochastic-gradient","title":"Parle: parallelizing stochastic gradient descent","date":"2017-07-03","arxiv_id":"1707.00424","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-scalable-inference-with-stochastic","title":"On Scalable Inference with Stochastic Gradient Descent","date":"2017-07-01","arxiv_id":"1707.00192","n_code_links":0,"syntology":null},{"paper":"/paper/spectrally-normalized-margin-bounds-for","slug":"spectrally-normalized-margin-bounds-for","title":"Spectrally-normalized margin bounds for neural networks","date":"2017-06-26","arxiv_id":"1706.08498","n_code_links":1,"syntology":null},{"paper":null,"slug":"collaborative-deep-learning-in-fixed-topology","title":"Collaborative Deep Learning in Fixed Topology Networks","date":"2017-06-23","arxiv_id":"1706.07880","n_code_links":0,"syntology":null},{"paper":null,"slug":"gradient-diversity-a-key-ingredient-for","title":"Gradient Diversity: a Key Ingredient for Scalable Distributed Learning","date":"2017-06-18","arxiv_id":"1706.05699","n_code_links":0,"syntology":null}],"record_sha256":"6eafb48d66bdcf2433716449e6267087f355e31d41ea7efe966a0e5a38a0237e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}