{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/16","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":16,"pages_in_order":21,"rows_per_page":100,"rows":[1501,1600],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/15","next":"/method/sgd/papers/17","papers":[{"paper":"/paper/communication-trade-offs-for-local-sgd-with","slug":"communication-trade-offs-for-local-sgd-with","title":"Communication trade-offs for Local-SGD with large step size","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"control-batch-size-and-learning-rate-to","title":"Control Batch Size and Learning Rate to Generalize Well: Theoretical and Empirical Evidence","date":"2019-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-meta-learning-via-minibatch","title":"Efficient Meta Learning via Minibatch Proximal Update","date":"2019-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/fast-and-accurate-stochastic-gradient","slug":"fast-and-accurate-stochastic-gradient","title":"Fast and Accurate Stochastic Gradient Estimation","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/optimal-sparsity-sensitive-bounds-for","slug":"optimal-sparsity-sensitive-bounds-for","title":"Optimal Sparsity-Sensitive Bounds for Distributed Mean Estimation","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/qsparse-local-sgd-distributed-sgd-with-1","slug":"qsparse-local-sgd-distributed-sgd-with-1","title":"Qsparse-local-SGD: Distributed SGD with Quantization, Sparsification and Local Computations","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"a-hybrid-approach-towards-two-stage-bengali","title":"A Hybrid Approach Towards Two Stage Bengali Question Classification Utilizing Smart Data Balancing Technique","date":"2019-11-30","arxiv_id":"1912.00127","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-heavy-tailed-theory-of-stochastic","title":"On the Heavy-Tailed Theory of Stochastic Gradient Descent for Deep Neural Networks","date":"2019-11-29","arxiv_id":"1912.00018","n_code_links":0,"syntology":null},{"paper":"/paper/deep-ordinal-classification-with-inequality","slug":"deep-ordinal-classification-with-inequality","title":"Non-parametric Uni-modality Constraints for Deep Ordinal Classification","date":"2019-11-25","arxiv_id":"1911.10720","n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-networks-learning-and-memorization","title":"Neural Networks Learning and Memorization with (almost) no Over-Parameterization","date":"2019-11-22","arxiv_id":"1911.09873","n_code_links":0,"syntology":null},{"paper":null,"slug":"parameter-free-locally-differentially-private","title":"Parameter-Free Locally Differentially Private Stochastic Subgradient Descent","date":"2019-11-21","arxiv_id":"1911.09564","n_code_links":0,"syntology":null},{"paper":null,"slug":"bayesian-interpretation-of-sgd-as-ito-process","title":"Bayesian interpretation of SGD as Ito process","date":"2019-11-20","arxiv_id":"1911.09011","n_code_links":0,"syntology":null},{"paper":"/paper/local-adaalter-communication-efficient","slug":"local-adaalter-communication-efficient","title":"Local AdaAlter: Communication-Efficient Stochastic Gradient Descent with Adaptive Learning Rates","date":"2019-11-20","arxiv_id":"1911.09030","n_code_links":1,"syntology":null},{"paper":null,"slug":"optimal-mini-batch-size-selection-for-fast","title":"Optimal Mini-Batch Size Selection for Fast Gradient Descent","date":"2019-11-15","arxiv_id":"1911.06459","n_code_links":0,"syntology":null},{"paper":null,"slug":"throughput-prediction-of-asynchronous-sgd-in","title":"Throughput Prediction of Asynchronous SGD in TensorFlow","date":"2019-11-12","arxiv_id":"1911.04650","n_code_links":0,"syntology":null},{"paper":null,"slug":"mindthestep-asyncpsgd-adaptive-asynchronous","title":"MindTheStep-AsyncPSGD: Adaptive Asynchronous Parallel Stochastic Gradient Descent","date":"2019-11-08","arxiv_id":"1911.03444","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-rule-for-gradient-estimator-selection-with","title":"A Rule for Gradient Estimator Selection, with an Application to Variational Inference","date":"2019-11-05","arxiv_id":"1911.01894","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforced-product-metadata-selection-for","title":"Reinforced Product Metadata Selection for Helpfulness Assessment of Customer Reviews","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"mixing-of-stochastic-accelerated-gradient","title":"Mixing of Stochastic Accelerated Gradient Descent","date":"2019-10-31","arxiv_id":"1910.14616","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-local-descent-methods","title":"On the Convergence of Local Descent Methods in Federated Learning","date":"2019-10-31","arxiv_id":"1910.14425","n_code_links":0,"syntology":null},{"paper":"/paper/local-sgd-with-periodic-averaging-tighter","slug":"local-sgd-with-periodic-averaging-tighter","title":"Local SGD with Periodic Averaging: Tighter Analysis and Adaptive Synchronization","date":"2019-10-30","arxiv_id":"1910.13598","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mmkamani7/LUPA-SGD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"lsh-sampling-breaks-the-computation-chicken","title":"Lsh-sampling Breaks the Computation Chicken-and-egg Loop in Adaptive Stochastic Gradient Estimation","date":"2019-10-30","arxiv_id":"1910.14162","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-stochastic-gradient-descent-with","title":"Online Stochastic Gradient Descent with Arbitrary Initialization Solves Non-smooth, Non-convex Phase Retrieval","date":"2019-10-28","arxiv_id":"1910.12837","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-geometric-interpretation-of-stochastic","title":"A geometric interpretation of stochastic gradient descent using diffusion metrics","date":"2019-10-27","arxiv_id":"1910.12194","n_code_links":0,"syntology":null},{"paper":null,"slug":"popsgd-decentralized-stochastic-gradient-1","title":"Asynchronous Decentralized SGD with Quantized and Local Updates","date":"2019-10-27","arxiv_id":"1910.12308","n_code_links":0,"syntology":null},{"paper":null,"slug":"sound-event-recognition-in-a-smart-city","title":"Sound Event Recognition in a Smart City Surveillance Context","date":"2019-10-27","arxiv_id":"1910.12369","n_code_links":0,"syntology":null},{"paper":null,"slug":"bias-variance-tradeoff-in-a-sliding-window","title":"Bias-Variance Tradeoff in a Sliding Window Implementation of the Stochastic Gradient Algorithm","date":"2019-10-25","arxiv_id":"1910.11868","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-practicality-of-stochastic-optimization","title":"The Practicality of Stochastic Optimization in Imaging Inverse Problems","date":"2019-10-22","arxiv_id":"1910.10100","n_code_links":0,"syntology":null},{"paper":null,"slug":"communication-efficient-decentralized","title":"Communication-Efficient Local Decentralized SGD Methods","date":"2019-10-21","arxiv_id":"1910.09126","n_code_links":0,"syntology":null},{"paper":null,"slug":"sparsification-as-a-remedy-for-staleness-in","title":"Sparsification as a Remedy for Staleness in Distributed Asynchronous SGD","date":"2019-10-21","arxiv_id":"1910.09466","n_code_links":0,"syntology":null},{"paper":null,"slug":"error-lower-bounds-of-constant-step-size","title":"Error Lower Bounds of Constant Step-size Stochastic Gradient Descent","date":"2019-10-18","arxiv_id":"1910.08212","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-the-convergence-of-sgd-through","title":"Improving the convergence of SGD through adaptive batch sizes","date":"2019-10-18","arxiv_id":"1910.08222","n_code_links":0,"syntology":null},{"paper":null,"slug":"interpreting-basis-path-set-in-neural","title":"Interpreting Basis Path Set in Neural Networks","date":"2019-10-18","arxiv_id":"1910.09402","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-learning-rate-selection-for-stochastic","title":"Robust Learning Rate Selection for Stochastic Optimization via Splitting Diagnostic","date":"2019-10-18","arxiv_id":"1910.08597","n_code_links":0,"syntology":null},{"paper":null,"slug":"why-bigger-is-not-always-better-on-finite-and","title":"Why bigger is not always better: on finite and infinite neural networks","date":"2019-10-17","arxiv_id":"1910.08013","n_code_links":0,"syntology":null},{"paper":"/paper/derivative-free-optimization-of-neural","slug":"derivative-free-optimization-of-neural","title":"Derivative-Free Optimization of Neural Networks using Local Search","date":"2019-10-15","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/dp-mac-the-differentially-private-method-of","slug":"dp-mac-the-differentially-private-method-of","title":"DP-MAC: The Differentially Private Method of Auxiliary Coordinates for Deep Learning","date":"2019-10-15","arxiv_id":"1910.06924","n_code_links":1,"syntology":null},{"paper":"/paper/decaying-momentum-helps-neural-network","slug":"decaying-momentum-helps-neural-network","title":"Demon: Improved Neural Network Training with Momentum Decay","date":"2019-10-11","arxiv_id":"1910.04952","n_code_links":2,"syntology":null},{"paper":"/paper/distributed-learning-of-deep-neural-networks","slug":"distributed-learning-of-deep-neural-networks","title":"Distributed Learning of Deep Neural Networks using Independent Subnet Training","date":"2019-10-04","arxiv_id":"1910.02120","n_code_links":2,"syntology":null},{"paper":null,"slug":"the-complexity-of-finding-stationary-points","title":"The Complexity of Finding Stationary Points with Stochastic Gradient Descent","date":"2019-10-04","arxiv_id":"1910.01845","n_code_links":0,"syntology":null},{"paper":"/paper/accelerating-deep-learning-by-focusing-on-the","slug":"accelerating-deep-learning-by-focusing-on-the","title":"Accelerating Deep Learning by Focusing on the Biggest Losers","date":"2019-10-02","arxiv_id":"1910.00762","n_code_links":2,"syntology":null},{"paper":null,"slug":"adaptive-activation-thresholding-dynamic","title":"Adaptive Activation Thresholding: Dynamic Routing Type Behavior for Interpretability in Convolutional Neural Networks","date":"2019-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"how-noise-affects-the-hessian-spectrum-in","title":"How noise affects the Hessian spectrum in overparameterized neural networks","date":"2019-10-01","arxiv_id":"1910.00195","n_code_links":0,"syntology":null},{"paper":"/paper/slowmo-improving-communication-efficient","slug":"slowmo-improving-communication-efficient","title":"SlowMo: Improving Communication-Efficient Distributed SGD with Slow Momentum","date":"2019-10-01","arxiv_id":"1910.00643","n_code_links":2,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":1,"phrase":"0 ran · 3 unverified","official":{"repos":["facebookresearch/fairscale"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":"/paper/small-steps-and-giant-leaps-minimal-newton-2","slug":"small-steps-and-giant-leaps-minimal-newton-2","title":"Small Steps and Giant Leaps: Minimal Newton Solvers for Deep Learning","date":"2019-10-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"distributed-sgd-generalizes-well-under","title":"Distributed SGD Generalizes Well Under Asynchrony","date":"2019-09-29","arxiv_id":"1909.13391","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-mean-field-theory-for-kernel-alignment-with","title":"A Mean-Field Theory for Kernel Alignment with Random Features in Generative and Discriminative Models","date":"2019-09-25","arxiv_id":"1909.11820","n_code_links":0,"syntology":null},{"paper":"/paper/attention-convolutional-binary-neural-tree","slug":"attention-convolutional-binary-neural-tree","title":"Attention Convolutional Binary Neural Tree for Fine-Grained Visual Categorization","date":"2019-09-25","arxiv_id":"1909.11378","n_code_links":2,"syntology":null},{"paper":null,"slug":"gap-aware-mitigation-of-gradient-staleness","title":"Gap Aware Mitigation of Gradient Staleness","date":"2019-09-24","arxiv_id":"1909.10802","n_code_links":0,"syntology":null},{"paper":null,"slug":"algorithm-for-training-neural-networks-on","title":"Algorithm for Training Neural Networks on Resistive Device Arrays","date":"2019-09-17","arxiv_id":"1909.07908","n_code_links":0,"syntology":null},{"paper":null,"slug":"finite-depth-and-width-corrections-to-the","title":"Finite Depth and Width Corrections to the Neural Tangent Kernel","date":"2019-09-13","arxiv_id":"1909.05989","n_code_links":0,"syntology":null},{"paper":"/paper/diffgrad-an-optimization-method-for","slug":"diffgrad-an-optimization-method-for","title":"diffGrad: An Optimization Method for Convolutional Neural Networks","date":"2019-09-12","arxiv_id":"1909.11015","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-error-feedback-framework-better-rates-for","title":"The Error-Feedback Framework: Better Rates for SGD with Delayed Gradients and Compressed Communication","date":"2019-09-11","arxiv_id":"1909.05350","n_code_links":0,"syntology":null},{"paper":null,"slug":"better-communication-complexity-for-local-sgd","title":"Tighter Theory for Local SGD on Identical and Heterogeneous Data","date":"2019-09-10","arxiv_id":"1909.04746","n_code_links":0,"syntology":null},{"paper":null,"slug":"byzantine-resilient-stochastic-gradient","title":"Byzantine-Resilient Stochastic Gradient Descent for Distributed Learning: A Lipschitz-Inspired Coordinate-wise Median Approach","date":"2019-09-10","arxiv_id":"1909.04532","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-stochastic-quasi-newton-method-with","title":"A Stochastic Quasi-Newton Method with Nesterov's Accelerated Gradient","date":"2019-09-09","arxiv_id":"1909.03621","n_code_links":0,"syntology":null},{"paper":"/paper/communication-censored-distributed-stochastic","slug":"communication-censored-distributed-stochastic","title":"Communication-Censored Distributed Stochastic Gradient Descent","date":"2019-09-09","arxiv_id":"1909.03631","n_code_links":1,"syntology":null},{"paper":null,"slug":"distributed-word2vec-using-graph-analytics","title":"Distributed Training of Embeddings using Graph Analytics","date":"2019-09-08","arxiv_id":"1909.03359","n_code_links":0,"syntology":null},{"paper":null,"slug":"decentralized-stochastic-gradient-tracking","title":"Decentralized Stochastic Gradient Tracking for Non-convex Empirical Risk Minimization","date":"2019-09-06","arxiv_id":"1909.02712","n_code_links":0,"syntology":null},{"paper":"/paper/freeanchor-learning-to-match-anchors-for","slug":"freeanchor-learning-to-match-anchors-for","title":"FreeAnchor: Learning to Match Anchors for Visual Object Detection","date":"2019-09-05","arxiv_id":"1909.02466","n_code_links":4,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhangxiaosong18/FreeAnchor"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"quasi-newton-optimization-methods-for-deep","title":"Quasi-Newton Optimization Methods For Deep Learning Applications","date":"2019-09-04","arxiv_id":"1909.01994","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-concert-planning-tool-for-independent","title":"A Concert-planning Tool for Independent Musicians by Machine Learning Models","date":"2019-08-29","arxiv_id":"1908.11200","n_code_links":0,"syntology":null},{"paper":"/paper/190807607","slug":"190807607","title":"Automatic and Simultaneous Adjustment of Learning Rate and Momentum for Stochastic Gradient Descent","date":"2019-08-20","arxiv_id":"1908.07607","n_code_links":1,"syntology":null},{"paper":"/paper/190807643","slug":"190807643","title":"AdaCliP: Adaptive Clipping for Private SGD","date":"2019-08-20","arxiv_id":"1908.07643","n_code_links":1,"syntology":null},{"paper":"/paper/multi-target-tracking-by-learning-from","slug":"multi-target-tracking-by-learning-from","title":"Multi Target Tracking by Learning from Generalized Graph Differences","date":"2019-08-19","arxiv_id":"1908.06646","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-better-generalization-bp-svrg-in","title":"Towards Better Generalization: BP-SVRG in Training Deep Neural Networks","date":"2019-08-18","arxiv_id":"1908.06395","n_code_links":0,"syntology":null},{"paper":"/paper/nuqsgd-improved-communication-efficiency-for","slug":"nuqsgd-improved-communication-efficiency-for","title":"NUQSGD: Improved Communication Efficiency for Data-parallel SGD via Nonuniform Quantization","date":"2019-08-16","arxiv_id":"1908.06077","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fartashf/nuqsgd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-the-convergence-of-adabound-and-its","slug":"on-the-convergence-of-adabound-and-its","title":"On the Convergence of AdaBound and its Connection to SGD","date":"2019-08-13","arxiv_id":"1908.04457","n_code_links":2,"syntology":null},{"paper":null,"slug":"taming-unbalanced-training-workloads-in-deep","title":"Taming Unbalanced Training Workloads in Deep Learning with Partial Collective Operations","date":"2019-08-12","arxiv_id":"1908.04207","n_code_links":0,"syntology":null},{"paper":"/paper/the-hsic-bottleneck-deep-learning-without","slug":"the-hsic-bottleneck-deep-learning-without","title":"The HSIC Bottleneck: Deep Learning without Back-Propagation","date":"2019-08-05","arxiv_id":"1908.01580","n_code_links":3,"syntology":{"ran":4,"of":16,"n_ran_checked":3,"n_instrument":1,"unverified":12,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 12 unverified","official":{"repos":["choasma/HSIC-Bottleneck"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":10,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"how-good-is-sgd-with-random-shuffling","title":"How Good is SGD with Random Shuffling?","date":"2019-07-31","arxiv_id":"1908.00045","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-gradient-boosting","title":"Deep Gradient Boosting -- Layer-wise Input Normalization of Neural Networks","date":"2019-07-29","arxiv_id":"1907.12608","n_code_links":0,"syntology":null},{"paper":null,"slug":"deam-accumulated-momentum-with-discriminative","title":"DEAM: Adaptive Momentum with Discriminative Weight for Stochastic Optimization","date":"2019-07-25","arxiv_id":"1907.11307","n_code_links":0,"syntology":null},{"paper":null,"slug":"hessian-based-analysis-of-sgd-for-deep-nets","title":"Hessian based analysis of SGD for Deep Nets: Dynamics and Generalization","date":"2019-07-24","arxiv_id":"1907.10732","n_code_links":0,"syntology":null},{"paper":"/paper/mix-and-match-an-optimistic-tree-search","slug":"mix-and-match-an-optimistic-tree-search","title":"Mix and Match: An Optimistic Tree-Search Approach for Learning Models from Mixture Distributions","date":"2019-07-23","arxiv_id":"1907.10154","n_code_links":1,"syntology":null},{"paper":null,"slug":"practical-newton-type-distributed-learning","title":"Practical Newton-Type Distributed Learning using Gradient Based Approximations","date":"2019-07-22","arxiv_id":"1907.09562","n_code_links":0,"syntology":null},{"paper":"/paper/speeding-up-iterative-closest-point-using","slug":"speeding-up-iterative-closest-point-using","title":"Speeding Up Iterative Closest Point Using Stochastic Gradient Descent","date":"2019-07-22","arxiv_id":"1907.09133","n_code_links":1,"syntology":null},{"paper":"/paper/post-synaptic-potential-regularization-has","slug":"post-synaptic-potential-regularization-has","title":"Post-synaptic potential regularization has potential","date":"2019-07-19","arxiv_id":"1907.08544","n_code_links":1,"syntology":null},{"paper":"/paper/sgd-momentum-optimizer-with-step-estimation","slug":"sgd-momentum-optimizer-with-step-estimation","title":"SGD momentum optimizer with step estimation by online parabola model","date":"2019-07-16","arxiv_id":"1907.07063","n_code_links":1,"syntology":null},{"paper":null,"slug":"amplifying-renyi-differential-privacy-via","title":"Amplifying Rényi Differential Privacy via Shuffling","date":"2019-07-11","arxiv_id":"1907.05156","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-highly-efficient-distributed-deep-learning","title":"A Highly Efficient Distributed Deep Learning System For Automatic Speech Recognition","date":"2019-07-10","arxiv_id":"1907.05701","n_code_links":0,"syntology":null},{"paper":"/paper/a-stochastic-first-order-method-for-ordered","slug":"a-stochastic-first-order-method-for-ordered","title":"Ordered SGD: A New Stochastic Optimization Framework for Empirical Risk Minimization","date":"2019-07-09","arxiv_id":"1907.04371","n_code_links":2,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["kenjikawaguchi/qSGD"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"unified-optimal-analysis-of-the-stochastic","title":"Unified Optimal Analysis of the (Stochastic) Gradient Method","date":"2019-07-09","arxiv_id":"1907.04232","n_code_links":0,"syntology":null},{"paper":null,"slug":"quantitative-w_1-convergence-of-langevin-like","title":"Stochastic Gradient and Langevin Processes","date":"2019-07-07","arxiv_id":"1907.03215","n_code_links":0,"syntology":null},{"paper":null,"slug":"next-generation-radiogenomics-sequencing-for","title":"Next Generation Radiogenomics Sequencing for Prediction of EGFR and KRAS Mutation Status in NSCLC Patients Using Multimodal Imaging and Machine Learning Approaches","date":"2019-07-03","arxiv_id":"1907.02121","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-symmetry-and-initialization-for-neural","title":"On Symmetry and Initialization for Neural Networks","date":"2019-07-01","arxiv_id":"1907.00560","n_code_links":0,"syntology":null},{"paper":null,"slug":"approximate-matrix-completion-based-on-cavity","title":"Approximate matrix completion based on cavity method","date":"2019-06-29","arxiv_id":"1907.00138","n_code_links":0,"syntology":null},{"paper":"/paper/dp-lssgd-a-stochastic-optimization-method-to","slug":"dp-lssgd-a-stochastic-optimization-method-to","title":"DP-LSSGD: A Stochastic Optimization Method to Lift the Utility in Privacy-Preserving ERM","date":"2019-06-28","arxiv_id":"1906.12056","n_code_links":1,"syntology":null},{"paper":null,"slug":"faster-distributed-deep-net-training","title":"Faster Distributed Deep Net Training: Computation and Communication Decoupled Stochastic Gradient Descent","date":"2019-06-28","arxiv_id":"1906.12043","n_code_links":0,"syntology":null},{"paper":null,"slug":"gradient-noise-convolution-gnc-smoothing-loss","title":"Gradient Noise Convolution (GNC): Smoothing Loss Function for Distributed Large-Batch SGD","date":"2019-06-26","arxiv_id":"1906.10822","n_code_links":0,"syntology":null},{"paper":"/paper/learning-data-augmentation-strategies-for","slug":"learning-data-augmentation-strategies-for","title":"Learning Data Augmentation Strategies for Object Detection","date":"2019-06-26","arxiv_id":"1906.11172","n_code_links":6,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tensorflow/tpu"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/first-exit-time-analysis-of-stochastic","slug":"first-exit-time-analysis-of-stochastic","title":"First Exit Time Analysis of Stochastic Gradient Descent Under Heavy-Tailed Gradient Noise","date":"2019-06-21","arxiv_id":"1906.09069","n_code_links":1,"syntology":null},{"paper":"/paper/data-cleansing-for-models-trained-with-sgd","slug":"data-cleansing-for-models-trained-with-sgd","title":"Data Cleansing for Models Trained with SGD","date":"2019-06-20","arxiv_id":"1906.08473","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sato9hara/sgd-influence"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/dynamics-of-stochastic-gradient-descent-for","slug":"dynamics-of-stochastic-gradient-descent-for","title":"Dynamics of stochastic gradient descent for two-layer neural networks in the teacher-student setup","date":"2019-06-18","arxiv_id":"1906.08632","n_code_links":3,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["sgoldt/nn2pp","sgoldt/pyscm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/the-multiplicative-noise-in-stochastic","slug":"the-multiplicative-noise-in-stochastic","title":"On the Noisy Gradient Descent that Generalizes as SGD","date":"2019-06-18","arxiv_id":"1906.07405","n_code_links":1,"syntology":null},{"paper":null,"slug":"remap-multi-layer-entropy-guided-pooling-of","title":"REMAP: Multi-layer entropy-guided pooling of dense CNN features for image retrieval","date":"2019-06-15","arxiv_id":"1906.06626","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-proximal-auc-maximization","title":"Stochastic Proximal AUC Maximization","date":"2019-06-14","arxiv_id":"1906.06053","n_code_links":0,"syntology":null},{"paper":null,"slug":"layered-sgd-a-decentralized-and-synchronous","title":"Layered SGD: A Decentralized and Synchronous SGD Algorithm for Scalable Deep Neural Network Training","date":"2019-06-13","arxiv_id":"1906.05936","n_code_links":0,"syntology":null},{"paper":"/paper/training-neural-networks-for-and-by","slug":"training-neural-networks-for-and-by","title":"Training Neural Networks for and by Interpolation","date":"2019-06-13","arxiv_id":"1906.05661","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["oval-group/ali-g"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"adass-adaptive-sample-selection-for-training","title":"ADASS: Adaptive Sample Selection for Training Acceleration","date":"2019-06-11","arxiv_id":"1906.04819","n_code_links":0,"syntology":null}],"record_sha256":"99c5146c3c8f2e64dee016ac63e034f841332c2979608b6b64c73c306b59cbc4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}