{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/17","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":17,"pages_in_order":21,"rows_per_page":100,"rows":[1601,1700],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/16","next":"/method/sgd/papers/18","papers":[{"paper":"/paper/adaptively-preconditioned-stochastic-gradient","slug":"adaptively-preconditioned-stochastic-gradient","title":"Adaptively Preconditioned Stochastic Gradient Langevin Dynamics","date":"2019-06-10","arxiv_id":"1906.04324","n_code_links":1,"syntology":null},{"paper":"/paper/stochastic-mirror-descent-on","slug":"stochastic-mirror-descent-on","title":"Stochastic Mirror Descent on Overparameterized Nonlinear Models: Convergence, Implicit Regularization, and Generalization","date":"2019-06-10","arxiv_id":"1906.03830","n_code_links":1,"syntology":null},{"paper":null,"slug":"making-asynchronous-stochastic-gradient","title":"Making Asynchronous Stochastic Gradient Descent Work for Transformers","date":"2019-06-08","arxiv_id":"1906.03496","n_code_links":0,"syntology":null},{"paper":"/paper/bad-global-minima-exist-and-sgd-can-reach","slug":"bad-global-minima-exist-and-sgd-can-reach","title":"Bad Global Minima Exist and SGD Can Reach Them","date":"2019-06-06","arxiv_id":"1906.02613","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chao1224/BadGlobalMinima"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"qsparse-local-sgd-distributed-sgd-with","title":"Qsparse-local-SGD: Distributed SGD with Quantization, Sparsification, and Local Computations","date":"2019-06-06","arxiv_id":"1906.02367","n_code_links":0,"syntology":null},{"paper":null,"slug":"binarized-collaborative-filtering-with","title":"Binarized Collaborative Filtering with Distilling Graph Convolutional Networks","date":"2019-06-05","arxiv_id":"1906.01829","n_code_links":0,"syntology":null},{"paper":"/paper/embedded-hyper-parameter-tuning-by-simulated","slug":"embedded-hyper-parameter-tuning-by-simulated","title":"Embedded hyper-parameter tuning by Simulated Annealing","date":"2019-06-04","arxiv_id":"1906.01504","n_code_links":2,"syntology":null},{"paper":"/paper/powersgd-practical-low-rank-gradient","slug":"powersgd-practical-low-rank-gradient","title":"PowerSGD: Practical Low-Rank Gradient Compression for Distributed Optimization","date":"2019-05-31","arxiv_id":"1905.13727","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["epfml/powersgd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"global-momentum-compression-for-sparse","title":"Global Momentum Compression for Sparse Communication in Distributed Learning","date":"2019-05-30","arxiv_id":"1905.12948","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-memory-based","title":"On the Convergence of Memory-Based Distributed SGD","date":"2019-05-30","arxiv_id":"1905.12960","n_code_links":0,"syntology":null},{"paper":null,"slug":"p3sgd-patient-privacy-preserving-sgd-for-1","title":"P3SGD: Patient Privacy Preserving SGD for Regularizing Deep CNNs in Pathological Image Classification","date":"2019-05-30","arxiv_id":"1905.12883","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerated-sparsified-sgd-with-error","title":"Accelerated Sparsified SGD with Error Feedback","date":"2019-05-29","arxiv_id":"1905.12224","n_code_links":0,"syntology":null},{"paper":null,"slug":"privacy-amplification-by-mixing-and-diffusion","title":"Privacy Amplification by Mixing and Diffusion Mechanisms","date":"2019-05-29","arxiv_id":"1905.12264","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-gram-gauss-newton-method-learning","title":"Gram-Gauss-Newton Method: Learning Overparameterized Neural Networks for Regression Problems","date":"2019-05-28","arxiv_id":"1905.11675","n_code_links":0,"syntology":null},{"paper":"/paper/sgd-on-neural-networks-learns-functions-of","slug":"sgd-on-neural-networks-learns-functions-of","title":"SGD on Neural Networks Learns Functions of Increasing Complexity","date":"2019-05-28","arxiv_id":"1905.11604","n_code_links":1,"syntology":null},{"paper":"/paper/communication-efficient-distributed-blockwise","slug":"communication-efficient-distributed-blockwise","title":"Communication-Efficient Distributed Blockwise Momentum SGD with Error-Feedback","date":"2019-05-27","arxiv_id":"1905.10936","n_code_links":1,"syntology":null},{"paper":null,"slug":"natural-compression-for-distributed-deep","title":"Natural Compression for Distributed Deep Learning","date":"2019-05-27","arxiv_id":"1905.10988","n_code_links":0,"syntology":null},{"paper":"/paper/stochastic-gradient-methods-with-layer-wise","slug":"stochastic-gradient-methods-with-layer-wise","title":"Stochastic Gradient Methods with Layer-wise Adaptive Moments for Training of Deep Networks","date":"2019-05-27","arxiv_id":"1905.11286","n_code_links":3,"syntology":{"ran":13,"of":16,"n_ran_checked":12,"n_instrument":1,"unverified":3,"pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/stochastic-shared-embeddings-data-driven","slug":"stochastic-shared-embeddings-data-driven","title":"Stochastic Shared Embeddings: Data-driven Regularization of Embedding Layers","date":"2019-05-25","arxiv_id":"1905.10630","n_code_links":3,"syntology":null},{"paper":null,"slug":"190512413","title":"VecHGrad for Solving Accurately Complex Tensor Decomposition","date":"2019-05-24","arxiv_id":"1905.12413","n_code_links":0,"syntology":null},{"paper":"/paper/painless-stochastic-gradient-interpolation","slug":"painless-stochastic-gradient-interpolation","title":"Painless Stochastic Gradient: Interpolation, Line-Search, and Convergence Rates","date":"2019-05-24","arxiv_id":"1905.09997","n_code_links":1,"syntology":null},{"paper":"/paper/matcha-speeding-up-decentralized-sgd-via","slug":"matcha-speeding-up-decentralized-sgd-via","title":"MATCHA: Speeding Up Decentralized SGD via Matching Decomposition Sampling","date":"2019-05-23","arxiv_id":"1905.09435","n_code_links":4,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/fine-grained-optimization-of-deep-neural","slug":"fine-grained-optimization-of-deep-neural","title":"Fine-grained Optimization of Deep Neural Networks","date":"2019-05-22","arxiv_id":"1905.09054","n_code_links":1,"syntology":null},{"paper":null,"slug":"time-smoothed-gradients-for-online","title":"Time-Smoothed Gradients for Online Forecasting","date":"2019-05-21","arxiv_id":"1905.08850","n_code_links":0,"syntology":null},{"paper":null,"slug":"shaping-the-learning-landscape-in-neural","title":"Shaping the learning landscape in neural networks around wide flat minima","date":"2019-05-20","arxiv_id":"1905.07833","n_code_links":0,"syntology":null},{"paper":"/paper/adaptively-truncating-backpropagation-through","slug":"adaptively-truncating-backpropagation-through","title":"Adaptively Truncating Backpropagation Through Time to Control Gradient Bias","date":"2019-05-17","arxiv_id":"1905.07473","n_code_links":1,"syntology":{"ran":18,"of":21,"n_ran_checked":18,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["aicherc/adaptive_tbptt"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/meta-reinforcement-learning-with-task","slug":"meta-reinforcement-learning-with-task","title":"Meta Reinforcement Learning with Task Embedding and Shared Policy","date":"2019-05-16","arxiv_id":"1905.06527","n_code_links":2,"syntology":null},{"paper":null,"slug":"doublesqueeze-parallel-stochastic-gradient","title":"DoubleSqueeze: Parallel Stochastic Gradient Descent with Double-Pass Error-Compensated Compression","date":"2019-05-15","arxiv_id":"1905.05957","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-computation-and-communication","title":"On the Computation and Communication Complexity of Parallel SGD with Dynamic Batch Sizes for Stochastic Non-Convex Optimization","date":"2019-05-10","arxiv_id":"1905.04346","n_code_links":0,"syntology":null},{"paper":null,"slug":"190503776","title":"The Effect of Network Width on Stochastic Gradient Descent and Generalization: an Empirical Study","date":"2019-05-09","arxiv_id":"1905.03776","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-linear-speedup-analysis-of","title":"On the Linear Speedup Analysis of Communication Efficient Momentum SGD for Distributed Non-Convex Optimization","date":"2019-05-09","arxiv_id":"1905.03817","n_code_links":0,"syntology":null},{"paper":"/paper/190503381","slug":"190503381","title":"AutoAssist: A Framework to Accelerate Training of Deep Neural Networks","date":"2019-05-08","arxiv_id":"1905.03381","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"190504374","title":"Fast and Robust Distributed Learning in High Dimension","date":"2019-05-05","arxiv_id":"1905.04374","n_code_links":0,"syntology":null},{"paper":"/paper/190503711","slug":"190503711","title":"Processing Megapixel Images with Deep Attention-Sampling Models","date":"2019-05-03","arxiv_id":"1905.03711","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["idiap/attention-sampling"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"a-unified-theory-of-adaptive-stochastic-1","title":"A unified theory of adaptive stochastic gradient descent as Bayesian filtering","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-walk-with-sgd-how-sgd-explores-regions-of","title":"A Walk with SGD: How SGD Explores Regions of Deep Network Loss?","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"asynchronous-sgd-without-gradient-delay-for","title":"Asynchronous SGD without gradient delay for efficient distributed training","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"dana-scalable-out-of-the-box-distributed-asgd","title":"DANA: Scalable Out-of-the-box Distributed ASGD Without Retuning","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"distributionally-robust-optimization-leads-to","title":"Distributionally Robust Optimization Leads to Better Generalization: on SGD and Beyond","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"g-sgd-optimizing-relu-neural-networks-in-its","title":"G-SGD: Optimizing ReLU Neural Networks in its Positively Scale-Invariant Space","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learn-from-neighbour-a-curriculum-that-train","title":"Learn From Neighbour: A Curriculum That Train Low Weighted Samples By Imitating","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"lsh-microbatches-for-stochastic-gradients-1","title":"LSH Microbatches for Stochastic Gradients: Value in Rearrangement","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-trajectory-of-stochastic-gradient","title":"On the Trajectory of Stochastic Gradient Descent in the Information Plane","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"online-hyperparameter-adaptation-via","title":"Online Hyperparameter Adaptation via Amortized Proximal Optimization","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"padam-closing-the-generalization-gap-of","title":"Padam: Closing the Generalization Gap of Adaptive Gradient Methods in Training Deep Neural Networks","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-anisotropic-noise-in-stochastic-gradient-1","title":"The Anisotropic Noise in Stochastic Gradient Descent: Its Behavior of Escaping from Minima and Regularization Effects","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"making-the-last-iterate-of-sgd-information","title":"Making the Last Iterate of SGD Information Theoretically Optimal","date":"2019-04-29","arxiv_id":"1904.12443","n_code_links":0,"syntology":null},{"paper":"/paper/the-step-decay-schedule-a-near-optimal","slug":"the-step-decay-schedule-a-near-optimal","title":"The Step Decay Schedule: A Near Optimal, Geometrically Decaying Learning Rate Procedure For Least Squares","date":"2019-04-29","arxiv_id":"1904.12838","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["D-X-Y/ResNeXt-DenseNet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/dynamic-mini-batch-sgd-for-elastic","slug":"dynamic-mini-batch-sgd-for-elastic","title":"Dynamic Mini-batch SGD for Elastic Distributed Training: Learning in the Limbo of Resources","date":"2019-04-26","arxiv_id":"1904.12043","n_code_links":2,"syntology":{"ran":13,"of":20,"n_ran_checked":13,"n_instrument":0,"unverified":7,"pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":null}},{"paper":"/paper/swalp-stochastic-weight-averaging-in-low","slug":"swalp-stochastic-weight-averaging-in-low","title":"SWALP : Stochastic Weight Averaging in Low-Precision Training","date":"2019-04-26","arxiv_id":"1904.11943","n_code_links":3,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["stevenygd/SWALP"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"communication-trade-offs-for-synchronized","title":"Communication trade-offs for synchronized distributed SGD with large step size","date":"2019-04-25","arxiv_id":"1904.11325","n_code_links":0,"syntology":null},{"paper":"/paper/reppoints-point-set-representation-for-object","slug":"reppoints-point-set-representation-for-object","title":"RepPoints: Point Set Representation for Object Detection","date":"2019-04-25","arxiv_id":"1904.11490","n_code_links":6,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/RepPoints"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"stability-and-optimization-error-of","title":"Stability and Optimization Error of Stochastic Gradient Descent for Pairwise Learning","date":"2019-04-25","arxiv_id":"1904.11316","n_code_links":0,"syntology":null},{"paper":null,"slug":"semi-cyclic-stochastic-gradient-descent","title":"Semi-Cyclic Stochastic Gradient Descent","date":"2019-04-23","arxiv_id":"1904.10120","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributed-deep-learning-strategies-for","title":"Distributed Deep Learning Strategies For Automatic Speech Recognition","date":"2019-04-10","arxiv_id":"1904.04956","n_code_links":0,"syntology":null},{"paper":"/paper/drop-an-octave-reducing-spatial-redundancy-in","slug":"drop-an-octave-reducing-spatial-redundancy-in","title":"Drop an Octave: Reducing Spatial Redundancy in Convolutional Neural Networks with Octave Convolution","date":"2019-04-10","arxiv_id":"1904.05049","n_code_links":28,"syntology":{"ran":24,"of":34,"n_ran_checked":11,"n_instrument":13,"unverified":10,"pointer_only":9,"phrase":"24 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 13 where Syntology's instrument failed) · 10 unverified","official":{"repos":["facebookresearch/OctConv"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/centripetal-sgd-for-pruning-very-deep","slug":"centripetal-sgd-for-pruning-very-deep","title":"Centripetal SGD for Pruning Very Deep Convolutional Networks with Complicated Structure","date":"2019-04-08","arxiv_id":"1904.03837","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-stochastic-interpretation-of-stochastic","title":"A Stochastic Interpretation of Stochastic Mirror Descent: Risk-Sensitive Optimality","date":"2019-04-03","arxiv_id":"1904.01855","n_code_links":0,"syntology":null},{"paper":null,"slug":"exponentially-convergent-stochastic-k-pca","title":"Exponentially convergent stochastic k-PCA without variance reduction","date":"2019-04-03","arxiv_id":"1904.01750","n_code_links":0,"syntology":null},{"paper":null,"slug":"normal-approximation-for-stochastic-gradient","title":"Normal Approximation for Stochastic Gradient Descent via Non-Asymptotic Rates of Martingale CLT","date":"2019-04-03","arxiv_id":"1904.02130","n_code_links":0,"syntology":null},{"paper":null,"slug":"lessons-from-building-acoustic-models-with-a","title":"Lessons from Building Acoustic Models with a Million Hours of Speech","date":"2019-04-02","arxiv_id":"1904.01624","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-stability-and-generalization-of","title":"On the Stability and Generalization of Learning with Kernel Activation Functions","date":"2019-03-28","arxiv_id":"1903.11990","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-competitive-and-discriminative","title":"Learning Competitive and Discriminative Reconstructions for Anomaly Detection","date":"2019-03-17","arxiv_id":"1903.07058","n_code_links":0,"syntology":null},{"paper":null,"slug":"inefficiency-of-k-fac-for-large-batch-size","title":"Inefficiency of K-FAC for Large Batch Size Training","date":"2019-03-14","arxiv_id":"1903.06237","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-distributed-hierarchical-sgd-algorithm-with","title":"A Distributed Hierarchical SGD Algorithm with Sparse Global Reduction","date":"2019-03-12","arxiv_id":"1903.05133","n_code_links":0,"syntology":null},{"paper":"/paper/communication-efficient-distributed-sgd-with","slug":"communication-efficient-distributed-sgd-with","title":"Communication-efficient distributed SGD with Sketching","date":"2019-03-12","arxiv_id":"1903.04488","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dhroth/sketchedsgd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"accelerating-minibatch-stochastic-gradient-1","title":"Accelerating Minibatch Stochastic Gradient Descent using Typicality Sampling","date":"2019-03-11","arxiv_id":"1903.04192","n_code_links":0,"syntology":null},{"paper":"/paper/partially-shuffling-the-training-data-to-1","slug":"partially-shuffling-the-training-data-to-1","title":"Partially Shuffling the Training Data to Improve Language Models","date":"2019-03-11","arxiv_id":"1903.04167","n_code_links":1,"syntology":null},{"paper":"/paper/fall-of-empires-breaking-byzantine-tolerant","slug":"fall-of-empires-breaking-byzantine-tolerant","title":"Fall of Empires: Breaking Byzantine-tolerant SGD by Inner Product Manipulation","date":"2019-03-10","arxiv_id":"1903.03936","n_code_links":4,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"time-delay-momentum-a-regularization","title":"Time-Delay Momentum: A Regularization Perspective on the Convergence and Generalization of Stochastic Momentum for Deep Learning","date":"2019-03-02","arxiv_id":"1903.00760","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributed-byzantine-tolerant-stochastic","title":"Distributed Byzantine Tolerant Stochastic Gradient Descent in the Era of Big Data","date":"2019-02-27","arxiv_id":"1902.10336","n_code_links":0,"syntology":null},{"paper":"/paper/equi-normalization-of-neural-networks","slug":"equi-normalization-of-neural-networks","title":"Equi-normalization of Neural Networks","date":"2019-02-27","arxiv_id":"1902.10416","n_code_links":1,"syntology":null},{"paper":"/paper/adaptive-gradient-methods-with-dynamic-bound","slug":"adaptive-gradient-methods-with-dynamic-bound","title":"Adaptive Gradient Methods with Dynamic Bound of Learning Rate","date":"2019-02-26","arxiv_id":"1902.09843","n_code_links":5,"syntology":null},{"paper":null,"slug":"beating-sgd-saturation-with-tail-averaging","title":"Beating SGD Saturation with Tail-Averaging and Minibatching","date":"2019-02-22","arxiv_id":"1902.08668","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-stochastic-gradient-descent-in","title":"Optimizing Stochastic Gradient Descent in Text Classification Based on Fine-Tuning Hyper-Parameters Approach. A Case Study on Automatic Classification of Global Terrorist Attacks","date":"2019-02-18","arxiv_id":"1902.06542","n_code_links":0,"syntology":null},{"paper":"/paper/multigrain-a-unified-image-embedding-for","slug":"multigrain-a-unified-image-embedding-for","title":"MultiGrain: a unified image embedding for classes and instances","date":"2019-02-14","arxiv_id":"1902.05509","n_code_links":3,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["facebookresearch/multigrain"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"scaling-limits-of-wide-neural-networks-with","title":"Scaling Limits of Wide Neural Networks with Weight Sharing: Gaussian Process Behavior, Gradient Independence, and Neural Tangent Kernel Derivation","date":"2019-02-13","arxiv_id":"1902.04760","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-escapes-saddle","title":"On Nonconvex Optimization for Machine Learning: Gradients, Stochasticity, and Saddle Points","date":"2019-02-13","arxiv_id":"1902.04811","n_code_links":0,"syntology":null},{"paper":"/paper/a-simple-baseline-for-bayesian-uncertainty-in","slug":"a-simple-baseline-for-bayesian-uncertainty-in","title":"A Simple Baseline for Bayesian Uncertainty in Deep Learning","date":"2019-02-07","arxiv_id":"1902.02476","n_code_links":8,"syntology":{"ran":11,"of":17,"n_ran_checked":10,"n_instrument":1,"unverified":6,"pointer_only":9,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["wjmaddox/swa_gaussian"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"a-scale-invariant-flatness-measure-for-deep","title":"A Scale Invariant Flatness Measure for Deep Network Minima","date":"2019-02-06","arxiv_id":"1902.02434","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-role-of-a-layer-in-deep-neural-networks-a","title":"The role of a layer in deep neural networks: a Gaussian Process perspective","date":"2019-02-06","arxiv_id":"1902.02354","n_code_links":0,"syntology":null},{"paper":null,"slug":"distribution-dependent-analysis-of-gibbs-erm","title":"Distribution-Dependent Analysis of Gibbs-ERM Principle","date":"2019-02-05","arxiv_id":"1902.01846","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-for-nonconvex","title":"Stochastic Gradient Descent for Nonconvex Learning without Bounded Gradient Assumptions","date":"2019-02-03","arxiv_id":"1902.00908","n_code_links":0,"syntology":null},{"paper":"/paper/asymmetric-valleys-beyond-sharp-and-flat","slug":"asymmetric-valleys-beyond-sharp-and-flat","title":"Asymmetric Valleys: Beyond Sharp and Flat Local Minima","date":"2019-02-02","arxiv_id":"1902.00744","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"uniform-in-time-weak-error-analysis-for","title":"Uniform-in-Time Weak Error Analysis for Stochastic Gradient Descent Algorithms via Diffusion Approximation","date":"2019-02-02","arxiv_id":"1902.00635","n_code_links":0,"syntology":null},{"paper":"/paper/compressing-gradient-optimizers-via-count","slug":"compressing-gradient-optimizers-via-count","title":"Compressing Gradient Optimizers via Count-Sketches","date":"2019-02-01","arxiv_id":"1902.00179","n_code_links":1,"syntology":null},{"paper":null,"slug":"sharp-analysis-for-nonconvex-sgd-escaping","title":"Sharp Analysis for Nonconvex SGD Escaping from Saddle Points","date":"2019-02-01","arxiv_id":"1902.00247","n_code_links":0,"syntology":null},{"paper":"/paper/error-feedback-fixes-signsgd-and-other","slug":"error-feedback-fixes-signsgd-and-other","title":"Error Feedback Fixes SignSGD and other Gradient Compression Schemes","date":"2019-01-28","arxiv_id":"1901.09847","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["epfml/error-feedback-SGD"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/iclr-reproducibility-challenge-report-padam","slug":"iclr-reproducibility-challenge-report-padam","title":"ICLR Reproducibility Challenge Report (Padam : Closing The Generalization Gap Of Adaptive Gradient Methods in Training Deep Neural Networks)","date":"2019-01-28","arxiv_id":"1901.09517","n_code_links":1,"syntology":null},{"paper":null,"slug":"99-of-parallel-optimization-is-inevitably-a","title":"99% of Distributed Optimization is a Waste of Time: The Issue and How to Fix it","date":"2019-01-27","arxiv_id":"1901.09437","n_code_links":0,"syntology":null},{"paper":"/paper/augment-your-batch-better-training-with","slug":"augment-your-batch-better-training-with","title":"Augment your batch: better training with larger batches","date":"2019-01-27","arxiv_id":"1901.09335","n_code_links":1,"syntology":null},{"paper":null,"slug":"sgd-general-analysis-and-improved-rates","title":"SGD: General Analysis and Improved Rates","date":"2019-01-27","arxiv_id":"1901.09401","n_code_links":0,"syntology":null},{"paper":null,"slug":"escaping-saddle-points-with-adaptive-gradient","title":"Escaping Saddle Points with Adaptive Gradient Methods","date":"2019-01-26","arxiv_id":"1901.09149","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalisation-dynamics-of-online-learning-in","title":"Generalisation dynamics of online learning in over-parameterised neural networks","date":"2019-01-25","arxiv_id":"1901.09085","n_code_links":0,"syntology":null},{"paper":"/paper/surrogate-losses-for-online-learning-of","slug":"surrogate-losses-for-online-learning-of","title":"Surrogate Losses for Online Learning of Stepsizes in Stochastic Non-Convex Optimization","date":"2019-01-25","arxiv_id":"1901.09068","n_code_links":1,"syntology":null},{"paper":null,"slug":"fitting-relus-via-sgd-and-quantized-sgd","title":"Fitting ReLUs via SGD and Quantized SGD","date":"2019-01-19","arxiv_id":"1901.06587","n_code_links":0,"syntology":null},{"paper":"/paper/a-tail-index-analysis-of-stochastic-gradient","slug":"a-tail-index-analysis-of-stochastic-gradient","title":"A Tail-Index Analysis of Stochastic Gradient Noise in Deep Neural Networks","date":"2019-01-18","arxiv_id":"1901.06053","n_code_links":1,"syntology":null},{"paper":null,"slug":"quasi-potential-as-an-implicit-regularizer","title":"Quasi-potential as an implicit regularizer for the loss function in the stochastic gradient descent","date":"2019-01-18","arxiv_id":"1901.06054","n_code_links":0,"syntology":null},{"paper":null,"slug":"recombination-of-artificial-neural-networks","title":"Recombination of Artificial Neural Networks","date":"2019-01-12","arxiv_id":"1901.03900","n_code_links":0,"syntology":null},{"paper":null,"slug":"quantized-epoch-sgd-for-communication","title":"Quantized Epoch-SGD for Communication-Efficient Distributed Learning","date":"2019-01-10","arxiv_id":"1901.03040","n_code_links":0,"syntology":null}],"record_sha256":"bc1843ddc893fb751e54690abca7fc3a8e1cf951ea35f4ae42b2f94529e54cef","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}