{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/11","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":11,"pages_in_order":21,"rows_per_page":100,"rows":[1001,1100],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/10","next":"/method/sgd/papers/12","papers":[{"paper":"/paper/activated-gradients-for-deep-neural-networks","slug":"activated-gradients-for-deep-neural-networks","title":"Activated Gradients for Deep Neural Networks","date":"2021-07-09","arxiv_id":"2107.04228","n_code_links":2,"syntology":null},{"paper":"/paper/rex-revisiting-budgeted-training-with-an","slug":"rex-revisiting-budgeted-training-with-an","title":"REX: Revisiting Budgeted Training with an Improved Schedule","date":"2021-07-09","arxiv_id":"2107.04197","n_code_links":1,"syntology":null},{"paper":"/paper/efficient-matrix-free-approximations-of","slug":"efficient-matrix-free-approximations-of","title":"M-FAC: Efficient Matrix-Free Approximations of Second-Order Information","date":"2021-07-07","arxiv_id":"2107.03356","n_code_links":2,"syntology":{"ran":7,"of":14,"n_ran_checked":4,"n_instrument":3,"unverified":7,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 4 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","official":{"repos":["IST-DASLab/M-FAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"adal-adaptive-gradient-transformation","title":"AdaL: Adaptive Gradient Transformation Contributes to Convergences and Generalizations","date":"2021-07-04","arxiv_id":"2107.01525","n_code_links":0,"syntology":null},{"paper":"/paper/exact-backpropagation-in-binary-weighted","slug":"exact-backpropagation-in-binary-weighted","title":"Exact Backpropagation in Binary Weighted Networks with Group Weight Transformations","date":"2021-07-03","arxiv_id":"2107.01400","n_code_links":1,"syntology":null},{"paper":null,"slug":"resist-layer-wise-decomposition-of-resnets","title":"ResIST: Layer-Wise Decomposition of ResNets for Distributed Training","date":"2021-07-02","arxiv_id":"2107.00961","n_code_links":0,"syntology":null},{"paper":null,"slug":"high-probability-bounds-for-non-convex","title":"High-probability Bounds for Non-Convex Stochastic Optimization with Heavy Tails","date":"2021-06-28","arxiv_id":"2106.14343","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-convergence-rate-of-sgd-s-final-iterate","title":"The Convergence Rate of SGD's Final Iterate: Analysis on Dimension Dependence","date":"2021-06-28","arxiv_id":"2106.14588","n_code_links":0,"syntology":null},{"paper":null,"slug":"low-precision-training-in-logarithmic-number","title":"LNS-Madam: Low-Precision Training in Logarithmic Number System using Multiplicative Weight Update","date":"2021-06-26","arxiv_id":"2106.13914","n_code_links":0,"syntology":null},{"paper":null,"slug":"implicit-gradient-alignment-in-distributed","title":"Implicit Gradient Alignment in Distributed and Federated Learning","date":"2021-06-25","arxiv_id":"2106.13897","n_code_links":0,"syntology":null},{"paper":null,"slug":"private-adaptive-gradient-methods-for-convex","title":"Private Adaptive Gradient Methods for Convex Optimization","date":"2021-06-25","arxiv_id":"2106.13756","n_code_links":0,"syntology":null},{"paper":null,"slug":"tighter-analysis-of-alternating-stochastic","title":"Tighter Analysis of Alternating Stochastic Gradient Method for Stochastic Nested Problems","date":"2021-06-25","arxiv_id":"2106.13781","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-clipping-for-federated-learning","title":"Understanding Clipping for Federated Learning: Convergence and Client-Level Differential Privacy","date":"2021-06-25","arxiv_id":"2106.13673","n_code_links":0,"syntology":null},{"paper":"/paper/numerical-influence-of-relu-0-on","slug":"numerical-influence-of-relu-0-on","title":"Numerical influence of ReLU'(0) on backpropagation","date":"2021-06-23","arxiv_id":"2106.12915","n_code_links":1,"syntology":null},{"paper":null,"slug":"stochastic-polyak-stepsize-with-a-moving","title":"Stochastic Polyak Stepsize with a Moving Target","date":"2021-06-22","arxiv_id":"2106.11851","n_code_links":0,"syntology":null},{"paper":"/paper/fedcm-federated-learning-with-client-level","slug":"fedcm-federated-learning-with-client-level","title":"FedCM: Federated Learning with Client-level Momentum","date":"2021-06-21","arxiv_id":"2106.10874","n_code_links":2,"syntology":null},{"paper":null,"slug":"how-do-adam-and-training-strategies-help-bnns","title":"How Do Adam and Training Strategies Help BNNs Optimization?","date":"2021-06-21","arxiv_id":"2106.11309","n_code_links":0,"syntology":null},{"paper":"/paper/open-set-label-noise-can-improve-robustness","slug":"open-set-label-noise-can-improve-robustness","title":"Open-set Label Noise Can Improve Robustness Against Inherent Label Noise","date":"2021-06-21","arxiv_id":"2106.10891","n_code_links":4,"syntology":{"ran":17,"of":29,"n_ran_checked":10,"n_instrument":7,"unverified":12,"pointer_only":29,"phrase":"17 ran (of which 5 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 7 where Syntology's instrument failed) · 12 unverified","official":{"repos":["hongxin001/ODNL"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":"/paper/multirate-training-of-neural-networks","slug":"multirate-training-of-neural-networks","title":"Multirate Training of Neural Networks","date":"2021-06-20","arxiv_id":"2106.10771","n_code_links":4,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tiffanyvlaar/multiratetrainingofnns"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"stem-a-stochastic-two-sided-momentum","title":"STEM: A Stochastic Two-Sided Momentum Algorithm Achieving Near-Optimal Sample and Communication Complexities for Federated Learning","date":"2021-06-19","arxiv_id":"2106.10435","n_code_links":0,"syntology":null},{"paper":null,"slug":"variational-prototype-learning-for-deep-face","title":"Variational Prototype Learning for Deep Face Recognition","date":"2021-06-19","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/large-scale-private-learning-via-low-rank","slug":"large-scale-private-learning-via-low-rank","title":"Large Scale Private Learning via Low-rank Reparametrization","date":"2021-06-17","arxiv_id":"2106.09352","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":1,"n_instrument":3,"unverified":4,"pointer_only":8,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["dayu11/Differentially-Private-Deep-Learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/locally-differentially-private-federated","slug":"locally-differentially-private-federated","title":"Private Federated Learning Without a Trusted Server: Optimal Algorithms for Convex Losses","date":"2021-06-17","arxiv_id":"2106.09779","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lowya/Locally-Differentially-Private-Federated-Learning","lowya/private-federated-learning-without-a-trusted-server"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"effective-evaluation-of-deep-active-learning","title":"Effective Evaluation of Deep Active Learning on Image Classification Tasks","date":"2021-06-16","arxiv_id":"2106.15324","n_code_links":0,"syntology":null},{"paper":null,"slug":"exponential-error-convergence-in-data","title":"Exponential Error Convergence in Data Classification with Optimized Random Features: Acceleration by Quantum Machine Learning","date":"2021-06-16","arxiv_id":"2106.09028","n_code_links":0,"syntology":null},{"paper":"/paper/robust-training-in-high-dimensions-via-block","slug":"robust-training-in-high-dimensions-via-block","title":"Robust Training in High Dimensions via Block Coordinate Geometric Median Descent","date":"2021-06-16","arxiv_id":"2106.08882","n_code_links":2,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["anishacharya/Optimization-Mavericks"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"simultaneous-training-of-partially-masked","title":"Masked Training of Neural Networks with Partial Gradients","date":"2021-06-16","arxiv_id":"2106.08895","n_code_links":0,"syntology":null},{"paper":"/paper/quantum-inspired-event-reconstruction-with","slug":"quantum-inspired-event-reconstruction-with","title":"Quantum-inspired event reconstruction with Tensor Networks: Matrix Product States","date":"2021-06-15","arxiv_id":"2106.08334","n_code_links":1,"syntology":null},{"paper":null,"slug":"revisiting-model-stitching-to-compare-neural","title":"Revisiting Model Stitching to Compare Neural Representations","date":"2021-06-14","arxiv_id":"2106.07682","n_code_links":0,"syntology":null},{"paper":"/paper/decreasing-scaling-transition-from-adaptive","slug":"decreasing-scaling-transition-from-adaptive","title":"A decreasing scaling transition scheme from Adam to SGD","date":"2021-06-12","arxiv_id":"2106.06749","n_code_links":2,"syntology":null},{"paper":"/paper/random-shuffling-beats-sgd-only-after-many","slug":"random-shuffling-beats-sgd-only-after-many","title":"Random Shuffling Beats SGD Only After Many Epochs on Ill-Conditioned Problems","date":"2021-06-12","arxiv_id":"2106.06880","n_code_links":1,"syntology":null},{"paper":null,"slug":"label-noise-sgd-provably-prefers-flat-global","title":"Label Noise SGD Provably Prefers Flat Global Minimizers","date":"2021-06-11","arxiv_id":"2106.06530","n_code_links":0,"syntology":null},{"paper":null,"slug":"communication-efficient-sgd-from-local-sgd-to","title":"Communication-efficient SGD: From Local SGD to One-Shot Averaging","date":"2021-06-09","arxiv_id":"2106.04759","n_code_links":0,"syntology":null},{"paper":null,"slug":"fractal-structure-and-generalization","title":"Fractal Structure and Generalization Properties of Stochastic Optimization Algorithms","date":"2021-06-09","arxiv_id":"2106.04881","n_code_links":0,"syntology":null},{"paper":"/paper/batch-normalization-orthogonalizes","slug":"batch-normalization-orthogonalizes","title":"Batch Normalization Orthogonalizes Representations in Deep Random Networks","date":"2021-06-07","arxiv_id":"2106.03970","n_code_links":1,"syntology":null},{"paper":null,"slug":"dynamics-of-stochastic-momentum-methods-on","title":"Dynamics of Stochastic Momentum Methods on Large-scale, Quadratic Models","date":"2021-06-07","arxiv_id":"2106.03696","n_code_links":0,"syntology":null},{"paper":"/paper/heavy-tails-in-sgd-and-compressibility-of","slug":"heavy-tails-in-sgd-and-compressibility-of","title":"Heavy Tails in SGD and Compressibility of Overparametrized Neural Networks","date":"2021-06-07","arxiv_id":"2106.03795","n_code_links":1,"syntology":null},{"paper":null,"slug":"vanishing-curvature-and-the-power-of-adaptive","title":"Vanishing Curvature and the Power of Adaptive Methods in Randomly Initialized Deep Networks","date":"2021-06-07","arxiv_id":"2106.03763","n_code_links":0,"syntology":null},{"paper":"/paper/fast-and-robust-online-inference-with","slug":"fast-and-robust-online-inference-with","title":"Fast and Robust Online Inference with Stochastic Gradient Descent via Random Scaling","date":"2021-06-06","arxiv_id":"2106.03156","n_code_links":1,"syntology":null},{"paper":null,"slug":"bandwidth-based-step-sizes-for-non-convex","title":"Bandwidth-based Step-Sizes for Non-Convex Stochastic Optimization","date":"2021-06-05","arxiv_id":"2106.02888","n_code_links":0,"syntology":null},{"paper":null,"slug":"escaping-saddle-points-faster-with-stochastic-1","title":"Escaping Saddle Points Faster with Stochastic Momentum","date":"2021-06-05","arxiv_id":"2106.02985","n_code_links":0,"syntology":null},{"paper":"/paper/debiasing-a-first-order-heuristic-for","slug":"debiasing-a-first-order-heuristic-for","title":"Debiasing a First-order Heuristic for Approximate Bi-level Optimization","date":"2021-06-04","arxiv_id":"2106.02487","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xingyousong/ufom"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-curves-for-sgd-on-structured","slug":"learning-curves-for-sgd-on-structured","title":"Learning Curves for SGD on Structured Features","date":"2021-06-04","arxiv_id":"2106.02713","n_code_links":1,"syntology":null},{"paper":"/paper/spreadgnn-serverless-multi-task-federated","slug":"spreadgnn-serverless-multi-task-federated","title":"SpreadGNN: Serverless Multi-task Federated Learning for Graph Neural Networks","date":"2021-06-04","arxiv_id":"2106.02743","n_code_links":1,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-with-noise-of-1","title":"Stochastic gradient descent with noise of machine learning type. Part II: Continuous time analysis","date":"2021-06-04","arxiv_id":"2106.02588","n_code_links":0,"syntology":null},{"paper":null,"slug":"continual-learning-in-deep-networks-an","title":"Continual Learning in Deep Networks: an Analysis of the Last Layer","date":"2021-06-03","arxiv_id":"2106.01834","n_code_links":0,"syntology":null},{"paper":"/paper/lrtuner-a-learning-rate-tuner-for-deep-neural","slug":"lrtuner-a-learning-rate-tuner-for-deep-neural","title":"LRTuner: A Learning Rate Tuner for Deep Neural Networks","date":"2021-05-30","arxiv_id":"2105.14526","n_code_links":2,"syntology":null},{"paper":"/paper/the-sobolev-regularization-effect-of","slug":"the-sobolev-regularization-effect-of","title":"On Linear Stability of SGD and Input-Smoothness of Neural Networks","date":"2021-05-27","arxiv_id":"2105.13462","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ChaoMa93/Sobolev-Reg-of-SGD"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"training-with-data-dependent-dynamic-learning","title":"Training With Data Dependent Dynamic Learning Rates","date":"2021-05-27","arxiv_id":"2105.13464","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-early-learning-regularization-to","title":"Using Early-Learning Regularization to Classify Real-World Noisy Data","date":"2021-05-27","arxiv_id":"2105.13244","n_code_links":0,"syntology":null},{"paper":null,"slug":"near-optimal-offline-and-streaming-algorithms","title":"Near-optimal Offline and Streaming Algorithms for Learning Non-Linear Dynamical Systems","date":"2021-05-24","arxiv_id":"2105.11558","n_code_links":0,"syntology":null},{"paper":null,"slug":"goals-gradient-only-approximations-for-line","title":"GOALS: Gradient-Only Approximations for Line Searches Towards Robust and Consistent Training of Deep Neural Networks","date":"2021-05-23","arxiv_id":"2105.10915","n_code_links":0,"syntology":null},{"paper":"/paper/angulargrad-a-new-optimization-technique-for","slug":"angulargrad-a-new-optimization-technique-for","title":"AngularGrad: A New Optimization Technique for Angular Convergence of Convolutional Neural Networks","date":"2021-05-21","arxiv_id":"2105.10190","n_code_links":3,"syntology":null},{"paper":null,"slug":"escaping-saddle-points-with-compressed-sgd","title":"Escaping Saddle Points with Compressed SGD","date":"2021-05-21","arxiv_id":"2105.10090","n_code_links":0,"syntology":null},{"paper":null,"slug":"last-iterate-convergence-of-sgd-for-least-1","title":"Last iterate convergence of SGD for Least-Squares in the Interpolation regime.","date":"2021-05-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"numerical-influence-of-relu-0-on-1","title":"Numerical influence of ReLU’(0) on backpropagation","date":"2021-05-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-generalization-of-neural-networks","title":"On the Generalization of Neural Networks Trained with SGD: Information-Theoretical Bounds and Implications","date":"2021-05-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"online-statistical-inference-for-parameters","title":"Online Statistical Inference for Parameters Estimation with Linear-Equality Constraints","date":"2021-05-21","arxiv_id":"2105.10315","n_code_links":0,"syntology":null},{"paper":null,"slug":"privacy-amplification-via-bernoulli-sampling","title":"Privacy Amplification Via Bernoulli Sampling","date":"2021-05-21","arxiv_id":"2105.10594","n_code_links":0,"syntology":null},{"paper":null,"slug":"properties-of-the-after-kernel","title":"Properties of the After Kernel","date":"2021-05-21","arxiv_id":"2105.10585","n_code_links":0,"syntology":null},{"paper":null,"slug":"logarithmic-landscape-and-power-law-escape","title":"Power-law escape rate of SGD","date":"2021-05-20","arxiv_id":"2105.09557","n_code_links":0,"syntology":null},{"paper":"/paper/towards-quantized-model-parallelism-for-graph","slug":"towards-quantized-model-parallelism-for-graph","title":"Towards Quantized Model Parallelism for Graph-Augmented MLPs Based on Gradient-Free ADMM Framework","date":"2021-05-20","arxiv_id":"2105.09837","n_code_links":1,"syntology":null},{"paper":null,"slug":"accelerating-gossip-sgd-with-periodic-global","title":"Accelerating Gossip SGD with Periodic Global Averaging","date":"2021-05-19","arxiv_id":"2105.09080","n_code_links":0,"syntology":null},{"paper":null,"slug":"removing-data-heterogeneity-influence","title":"Removing Data Heterogeneity Influence Enhances Network Topology Dependence of Decentralized SGD","date":"2021-05-17","arxiv_id":"2105.08023","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-qa-fast-schema-guided-dialogue-state","title":"SGD-QA: Fast Schema-Guided Dialogue State Tracking for Unseen Services","date":"2021-05-17","arxiv_id":"2105.08049","n_code_links":0,"syntology":null},{"paper":null,"slug":"drill-the-cork-of-information-bottleneck-by","title":"Drill the Cork of Information Bottleneck by Inputting the Most Important Data","date":"2021-05-15","arxiv_id":"2105.07181","n_code_links":0,"syntology":null},{"paper":null,"slug":"why-does-multi-epoch-training-help","title":"Why Does Multi-Epoch Training Help?","date":"2021-05-13","arxiv_id":"2105.06015","n_code_links":0,"syntology":null},{"paper":"/paper/evading-the-simplicity-bias-training-a","slug":"evading-the-simplicity-bias-training-a","title":"Evading the Simplicity Bias: Training a Diverse Set of Models Discovers Solutions with Superior OOD Generalization","date":"2021-05-12","arxiv_id":"2105.05612","n_code_links":1,"syntology":null},{"paper":null,"slug":"tensor-programs-iib-architectural","title":"Tensor Programs IIb: Architectural Universality of Neural Tangent Kernel Training Dynamics","date":"2021-05-08","arxiv_id":"2105.03703","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-long-range-memory-effects-in","title":"Understanding Short-Range Memory Effects in Deep Neural Networks","date":"2021-05-05","arxiv_id":"2105.02062","n_code_links":0,"syntology":null},{"paper":null,"slug":"information-complexity-and-generalization","title":"Information Complexity and Generalization Bounds","date":"2021-05-04","arxiv_id":"2105.01747","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-with-noise-of","title":"Stochastic gradient descent with noise of machine learning type. Part I: Discrete time analysis","date":"2021-05-04","arxiv_id":"2105.01650","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-perceptual-distortion-reduction-framework","title":"A Perceptual Distortion Reduction Framework: Towards Generating Adversarial Examples with High Perceptual Quality and Attack Success Rate","date":"2021-05-01","arxiv_id":"2105.00278","n_code_links":0,"syntology":null},{"paper":null,"slug":"one-pass-stochastic-gradient-descent-in","title":"One-pass Stochastic Gradient Descent in Overparametrized Two-layer Neural Networks","date":"2021-05-01","arxiv_id":"2105.00262","n_code_links":0,"syntology":null},{"paper":null,"slug":"studying-the-consistency-and-composability-of","title":"Studying the Consistency and Composability of Lottery Ticket Pruning Masks","date":"2021-04-30","arxiv_id":"2104.14753","n_code_links":0,"syntology":null},{"paper":"/paper/nuqsgd-provably-communication-efficient-data","slug":"nuqsgd-provably-communication-efficient-data","title":"NUQSGD: Provably Communication-efficient Data-parallel SGD via Nonuniform Quantization","date":"2021-04-28","arxiv_id":"2104.13818","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-comparison-between-cyclic-sampling-and","title":"Improved Analysis and Rates for Variance Reduction under Without-replacement Sampling Orders","date":"2021-04-25","arxiv_id":"2104.12112","n_code_links":0,"syntology":null},{"paper":"/paper/decentlam-decentralized-momentum-sgd-for","slug":"decentlam-decentralized-momentum-sgd-for","title":"DecentLaM: Decentralized Momentum SGD for Large-batch Deep Training","date":"2021-04-24","arxiv_id":"2104.11981","n_code_links":1,"syntology":null},{"paper":null,"slug":"partitioning-sparse-deep-neural-networks-for","title":"Partitioning sparse deep neural networks for scalable training and inference","date":"2021-04-23","arxiv_id":"2104.11805","n_code_links":0,"syntology":null},{"paper":null,"slug":"metricopt-learning-to-optimize-black-box","title":"MetricOpt: Learning to Optimize Black-Box Evaluation Metrics","date":"2021-04-21","arxiv_id":"2104.10631","n_code_links":0,"syntology":null},{"paper":null,"slug":"random-reshuffling-with-variance-reduction","title":"Random Reshuffling with Variance Reduction: New Analysis and Better Rates","date":"2021-04-19","arxiv_id":"2104.09342","n_code_links":0,"syntology":null},{"paper":"/paper/a-method-to-reveal-speaker-identity-in","slug":"a-method-to-reveal-speaker-identity-in","title":"A Method to Reveal Speaker Identity in Distributed ASR Training, and How to Counter It","date":"2021-04-15","arxiv_id":"2104.07815","n_code_links":1,"syntology":null},{"paper":null,"slug":"d-cliques-compensating-noniidness-in","title":"D-Cliques: Compensating for Data Heterogeneity with Topology in Decentralized Federated Learning","date":"2021-04-15","arxiv_id":"2104.07365","n_code_links":0,"syntology":null},{"paper":"/paper/1-bit-lamb-communication-efficient-large","slug":"1-bit-lamb-communication-efficient-large","title":"1-bit LAMB: Communication Efficient Large-Scale Large-Batch Training with LAMB's Convergence Speed","date":"2021-04-13","arxiv_id":"2104.06069","n_code_links":1,"syntology":null},{"paper":"/paper/sample-based-and-feature-based-federated","slug":"sample-based-and-feature-based-federated","title":"Sample-based and Feature-based Federated Learning for Unconstrained and Constrained Nonconvex Optimization via Mini-batch SSCA","date":"2021-04-13","arxiv_id":"2104.06011","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-based-chinese-text-classification-for","title":"BERT-based Chinese Text Classification for Emergency Domain with a Novel Loss Function","date":"2021-04-09","arxiv_id":"2104.04197","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-proof-of-convergence-for-stochastic","title":"A proof of convergence for stochastic gradient descent in the training of artificial neural networks with ReLU activation for constant target functions","date":"2021-04-01","arxiv_id":"2104.00277","n_code_links":0,"syntology":null},{"paper":"/paper/empirically-explaining-sgd-from-a-line-search","slug":"empirically-explaining-sgd-from-a-line-search","title":"Empirically explaining SGD from a line search perspective","date":"2021-03-31","arxiv_id":"2103.17132","n_code_links":1,"syntology":null},{"paper":"/paper/positive-negative-momentum-manipulating","slug":"positive-negative-momentum-manipulating","title":"Positive-Negative Momentum: Manipulating Stochastic Gradient Noise to Improve Generalization","date":"2021-03-31","arxiv_id":"2103.17182","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["zeke-xie/Positive-Negative-Momentum"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"research-of-damped-newton-stochastic-gradient","title":"Research of Damped Newton Stochastic Gradient Descent Method for Neural Network Training","date":"2021-03-31","arxiv_id":"2103.16764","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploiting-adam-like-optimization-algorithms","title":"Exploiting Adam-like Optimization Algorithms to Improve the Performance of Convolutional Neural Networks","date":"2021-03-26","arxiv_id":"2103.14689","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-importance-sampling-for-finite-sum-1","title":"Adaptive Importance Sampling for Finite-Sum Optimization and Sampling with Decreasing Step-Sizes","date":"2021-03-23","arxiv_id":"2103.12243","n_code_links":0,"syntology":null},{"paper":null,"slug":"benign-overfitting-of-constant-stepsize-sgd","title":"Benign Overfitting of Constant-Stepsize SGD for Linear Regression","date":"2021-03-23","arxiv_id":"2103.12692","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-reweighted-gradient-descent","title":"Stochastic Reweighted Gradient Descent","date":"2021-03-23","arxiv_id":"2103.12293","n_code_links":0,"syntology":null},{"paper":"/paper/progressivespinalnet-architecture-for-fc","slug":"progressivespinalnet-architecture-for-fc","title":"ProgressiveSpinalNet architecture for FC layers","date":"2021-03-21","arxiv_id":"2103.11373","n_code_links":1,"syntology":null},{"paper":"/paper/datalens-scalable-privacy-preserving-training","slug":"datalens-scalable-privacy-preserving-training","title":"DataLens: Scalable Privacy Preserving Training via Gradient Compression and Aggregation","date":"2021-03-20","arxiv_id":"2103.11109","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-deep-learning-theory-for-neural-networks","title":"A deep learning theory for neural networks grounded in physics","date":"2021-03-18","arxiv_id":"2103.09985","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributed-deep-learning-using-volunteer","title":"Distributed Deep Learning Using Volunteer Computing-Like Paradigm","date":"2021-03-16","arxiv_id":"2103.08894","n_code_links":0,"syntology":null},{"paper":null,"slug":"hebbian-semi-supervised-learning-in-a-sample","title":"Hebbian Semi-Supervised Learning in a Sample Efficiency Setting","date":"2021-03-16","arxiv_id":"2103.09002","n_code_links":0,"syntology":null},{"paper":"/paper/repurposing-pretrained-models-for-robust-out-1","slug":"repurposing-pretrained-models-for-robust-out-1","title":"Repurposing Pretrained Models for Robust Out-of-domain Few-Shot Learning","date":"2021-03-16","arxiv_id":"2103.09027","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["NamyeongK/USA_UFGSM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"6f07c7cd7b555b6ae96e33e059a26ba309b205257d45735582cb5d7ef631f435","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}