{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/5","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":5,"pages_in_order":21,"rows_per_page":100,"rows":[401,500],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/4","next":"/method/sgd/papers/6","papers":[{"paper":"/paper/a-quadratic-synchronization-rule-for","slug":"a-quadratic-synchronization-rule-for","title":"A Quadratic Synchronization Rule for Distributed Deep Learning","date":"2023-10-22","arxiv_id":"2310.14423","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hmgxr128/qsr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-hyperparameter-agnostic-dnn-training","title":"Towards Hyperparameter-Agnostic DNN Training via Dynamical System Insights","date":"2023-10-21","arxiv_id":"2310.13901","n_code_links":0,"syntology":null},{"paper":null,"slug":"demystifying-the-myths-and-legends-of","title":"Demystifying the Myths and Legends of Nonconvex Convergence of SGD","date":"2023-10-19","arxiv_id":"2310.12969","n_code_links":0,"syntology":null},{"paper":null,"slug":"laser-linear-compression-in-wireless","title":"LASER: Linear Compression in Wireless Distributed Optimization","date":"2023-10-19","arxiv_id":"2310.13033","n_code_links":0,"syntology":null},{"paper":null,"slug":"jorge-approximate-preconditioning-for-gpu","title":"Jorge: Approximate Preconditioning for GPU-efficient Second-order Optimization","date":"2023-10-18","arxiv_id":"2310.12298","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-generate-parameters-of-convnets","slug":"learning-to-generate-parameters-of-convnets","title":"Learning to Generate Parameters of ConvNets for Unseen Image Data","date":"2023-10-18","arxiv_id":"2310.11862","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-automatic-learning-rate-schedule-algorithm","title":"An Automatic Learning Rate Schedule Algorithm for Achieving Faster Convergence and Steeper Descent","date":"2023-10-17","arxiv_id":"2310.11291","n_code_links":0,"syntology":null},{"paper":null,"slug":"butterfly-effects-of-sgd-noise-error","title":"Butterfly Effects of SGD Noise: Error Amplification in Behavior Cloning and Autoregression","date":"2023-10-17","arxiv_id":"2310.11428","n_code_links":0,"syntology":null},{"paper":null,"slug":"resampling-stochastic-gradient-descent","title":"Resampling Stochastic Gradient Descent Cheaply for Efficient Uncertainty Quantification","date":"2023-10-17","arxiv_id":"2310.11065","n_code_links":0,"syntology":null},{"paper":null,"slug":"contextual-data-augmentation-for-task","title":"Contextual Data Augmentation for Task-Oriented Dialog Systems","date":"2023-10-16","arxiv_id":"2310.10380","n_code_links":0,"syntology":null},{"paper":null,"slug":"adam-family-methods-with-decoupled-weight","title":"Adam-family Methods with Decoupled Weight Decay in Deep Learning","date":"2023-10-13","arxiv_id":"2310.08858","n_code_links":0,"syntology":null},{"paper":null,"slug":"asynchronous-federated-learning-with-2","title":"Asynchronous Federated Learning with Incentive Mechanism Based on Contract Theory","date":"2023-10-10","arxiv_id":"2310.06448","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamical-versus-bayesian-phase-transitions","title":"Dynamical versus Bayesian Phase Transitions in a Toy Model of Superposition","date":"2023-10-10","arxiv_id":"2310.06301","n_code_links":0,"syntology":null},{"paper":null,"slug":"augmenting-vision-based-human-pose-estimation","title":"Augmenting Vision-Based Human Pose Estimation with Rotation Matrix","date":"2023-10-09","arxiv_id":"2310.06068","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-predicting-and-better-resolving-1","slug":"understanding-predicting-and-better-resolving-1","title":"Understanding, Predicting and Better Resolving Q-Value Divergence in Offline-RL","date":"2023-10-06","arxiv_id":"2310.04411","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 4 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yueyang130/seem"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/why-do-we-need-weight-decay-in-modern-deep","slug":"why-do-we-need-weight-decay-in-modern-deep","title":"Why Do We Need Weight Decay in Modern Deep Learning?","date":"2023-10-06","arxiv_id":"2310.04415","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tml-epfl/why-weight-decay"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"high-dimensional-sgd-aligns-with-emerging","title":"Spectral alignment of stochastic gradient descent for high-dimensional classification tasks","date":"2023-10-04","arxiv_id":"2310.03010","n_code_links":0,"syntology":null},{"paper":null,"slug":"hoeffding-s-inequality-for-markov-chains","title":"Hoeffding's Inequality for Markov Chains under Generalized Concentrability Condition","date":"2023-10-04","arxiv_id":"2310.02941","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-simple-connection-from-loss-flatness-to","title":"A simple connection from loss flatness to compressed neural representations","date":"2023-10-03","arxiv_id":"2310.01770","n_code_links":0,"syntology":null},{"paper":"/paper/chunking-forgetting-matters-in-continual","slug":"chunking-forgetting-matters-in-continual","title":"Chunking: Continual Learning is not just about Distribution Shift","date":"2023-10-03","arxiv_id":"2310.02206","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tlee43/chunking-setting"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-the-parallel-complexity-of-multilevel","title":"On the Parallel Complexity of Multilevel Monte Carlo in Stochastic Gradient Descent","date":"2023-10-03","arxiv_id":"2310.02402","n_code_links":0,"syntology":null},{"paper":null,"slug":"symmetric-single-index-learning","title":"Symmetric Single Index Learning","date":"2023-10-03","arxiv_id":"2310.02117","n_code_links":0,"syntology":null},{"paper":null,"slug":"batch-less-stochastic-gradient-descent-for","title":"Batch-less stochastic gradient descent for compressive learning of deep regularization for image denoising","date":"2023-10-02","arxiv_id":"2310.03085","n_code_links":0,"syntology":null},{"paper":"/paper/improving-dialogue-management-quality","slug":"improving-dialogue-management-quality","title":"Improving Dialogue Management: Quality Datasets vs Models","date":"2023-10-02","arxiv_id":"2310.01339","n_code_links":1,"syntology":null},{"paper":null,"slug":"stability-and-generalization-for-minibatch","title":"Stability and Generalization for Minibatch SGD and Local SGD","date":"2023-10-02","arxiv_id":"2310.01139","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-noise-geometry-of-stochastic-gradient","title":"A Theoretical Analysis of Noise Geometry in Stochastic Gradient Descent","date":"2023-10-01","arxiv_id":"2310.00692","n_code_links":0,"syntology":null},{"paper":"/paper/on-memorization-and-privacy-risks-of","slug":"on-memorization-and-privacy-risks-of","title":"On Memorization and Privacy Risks of Sharpness Aware Minimization","date":"2023-09-30","arxiv_id":"2310.00488","n_code_links":0,"syntology":{"ran":7,"of":9,"n_ran_checked":6,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"robust-stochastic-optimization-via-gradient","title":"Robust Stochastic Optimization via Gradient Quantile Clipping","date":"2023-09-29","arxiv_id":"2309.17316","n_code_links":0,"syntology":null},{"paper":null,"slug":"autoencoding-tree-for-city-generation-and","title":"AutoEncoding Tree for City Generation and Applications","date":"2023-09-27","arxiv_id":"2309.15941","n_code_links":0,"syntology":null},{"paper":null,"slug":"fixing-the-ntk-from-neural-network","title":"Fixing the NTK: From Neural Network Linearizations to Exact Convex Programs","date":"2023-09-26","arxiv_id":"2309.15096","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-finds-then-tunes-features-in-two-layer","title":"SGD Finds then Tunes Features in Two-Layer Neural Networks with near-Optimal Sample Complexity: A Case Study in the XOR problem","date":"2023-09-26","arxiv_id":"2309.15111","n_code_links":0,"syntology":null},{"paper":null,"slug":"grounding-description-driven-dialogue-state","title":"Grounding Description-Driven Dialogue State Trackers with Knowledge-Seeking Turns","date":"2023-09-23","arxiv_id":"2309.13448","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-guide-through-the-zoo-of-biased-sgd","title":"A Guide Through the Zoo of Biased SGD","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/approximate-heavy-tails-in-offline-multi-pass","slug":"approximate-heavy-tails-in-offline-multi-pass","title":"Approximate Heavy Tails in Offline (Multi-Pass) Stochastic Gradient Descent","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"augmented-memory-replay-based-continual","title":"Augmented Memory Replay-based Continual Learning Approaches for Network Intrusion Detection","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/automatic-clipping-differentially-private","slug":"automatic-clipping-differentially-private","title":"Automatic Clipping: Differentially Private Deep Learning Made Easier and Stronger","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/bayestune-bayesian-sparse-deep-model-fine","slug":"bayestune-bayesian-sparse-deep-model-fine","title":"BayesTune: Bayesian Sparse Deep Model Fine-tuning","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/beyond-exponential-graph-communication","slug":"beyond-exponential-graph-communication","title":"Beyond Exponential Graph: Communication-Efficient Topologies for Decentralized Learning via Finite-time Convergence","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/differentially-private-image-classification-1","slug":"differentially-private-image-classification-1","title":"Differentially Private Image Classification by Learning Priors from Random Processes","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"easy-bayesian-transfer-learning-with","title":"Easy Bayesian Transfer Learning with Informative Priors","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/federated-learning-with-client-subsampling","slug":"federated-learning-with-client-subsampling","title":"Federated Learning with Client Subsampling, Data Heterogeneity, and Unbounded Smoothness: A New Algorithm and Lower Bounds","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"global-convergence-analysis-of-local-sgd-for","title":"Global Convergence Analysis of Local SGD for Two-layer Neural Network without Overparameterization","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"implicit-bias-of-stochastic-gradient-descent","title":"Implicit Bias of (Stochastic) Gradient Descent for Rank-1 Linear Neural Network","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/kakurenbo-adaptively-hiding-samples-in-deep","slug":"kakurenbo-adaptively-hiding-samples-in-deep","title":"KAKURENBO: Adaptively Hiding Samples in Deep Neural Network Training","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"mean-field-langevin-dynamics-time-space","title":"Mean-field Langevin dynamics: Time-space discretization, stochastic gradient, and variance reduction","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"s-gd-over-diagonal-linear-networks-implicit-1","title":"(S)GD over Diagonal Linear Networks: Implicit bias, Large Stepsizes and Edge of Stability","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"preconditioned-federated-learning","title":"Preconditioned Federated Learning","date":"2023-09-20","arxiv_id":"2309.11378","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-different-regimes-of-stochastic","slug":"on-the-different-regimes-of-stochastic","title":"On the different regimes of Stochastic Gradient Descent","date":"2023-09-19","arxiv_id":"2309.10688","n_code_links":1,"syntology":{"ran":9,"of":17,"n_ran_checked":9,"n_instrument":0,"unverified":8,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","official":{"repos":["pcsl-epfl/regimes_of_sgd"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"global-convergence-of-sgd-for-logistic-loss","title":"Global Convergence of SGD For Logistic Loss on Two Layer Neural Nets","date":"2023-09-17","arxiv_id":"2309.09258","n_code_links":0,"syntology":null},{"paper":"/paper/stochastic-gradient-descent-like-relaxation","slug":"stochastic-gradient-descent-like-relaxation","title":"Stochastic Gradient Descent-like relaxation is equivalent to Metropolis dynamics in discrete optimization and inference problems","date":"2023-09-11","arxiv_id":"2309.05337","n_code_links":1,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-outperforms","title":"Stochastic Gradient Descent outperforms Gradient Descent in recovering a high-dimensional signal in a glassy energy landscape","date":"2023-09-09","arxiv_id":"2309.04788","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-analysis-of-decentralized-asgd","title":"Convergence Analysis of Decentralized ASGD","date":"2023-09-07","arxiv_id":"2309.03754","n_code_links":0,"syntology":null},{"paper":"/paper/adaplus-integrating-nesterov-momentum-and","slug":"adaplus-integrating-nesterov-momentum-and","title":"AdaPlus: Integrating Nesterov Momentum and Precise Stepsize Adjustment on AdamW Basis","date":"2023-09-05","arxiv_id":"2309.01966","n_code_links":1,"syntology":null},{"paper":"/paper/asymmetric-momentum-a-rethinking-of-gradient","slug":"asymmetric-momentum-a-rethinking-of-gradient","title":"Asymmetric Momentum: A Rethinking of Gradient Descent","date":"2023-09-05","arxiv_id":"2309.02130","n_code_links":1,"syntology":null},{"paper":null,"slug":"corgi-2-a-hybrid-offline-online-approach-to","title":"Corgi^2: A Hybrid Offline-Online Approach To Storage-Aware Data Shuffling For SGD","date":"2023-09-04","arxiv_id":"2309.01640","n_code_links":0,"syntology":null},{"paper":null,"slug":"abs-sgd-a-delayed-synchronous-stochastic","title":"ABS-SGD: A Delayed Synchronous Stochastic Gradient Descent Algorithm with Adaptive Batch Size for Heterogeneous GPU Clusters","date":"2023-08-29","arxiv_id":"2308.15164","n_code_links":0,"syntology":null},{"paper":null,"slug":"bias-aware-minimisation-understanding-and","title":"Bias-Aware Minimisation: Understanding and Mitigating Estimator Bias in Private SGD","date":"2023-08-23","arxiv_id":"2308.12018","n_code_links":0,"syntology":null},{"paper":null,"slug":"extended-linear-regression-a-kalman-filter","title":"Extended Linear Regression: A Kalman Filter Approach for Minimizing Loss via Area Under the Curve","date":"2023-08-23","arxiv_id":"2308.12280","n_code_links":0,"syntology":null},{"paper":null,"slug":"when-minibatch-sgd-meets-splitfed-learning","title":"When MiniBatch SGD Meets SplitFed Learning:Convergence Analysis and Performance Evaluation","date":"2023-08-23","arxiv_id":"2308.11953","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-understanding-the-generalizability-of","title":"Towards Understanding the Generalizability of Delayed Stochastic Gradient Descent","date":"2023-08-18","arxiv_id":"2308.09430","n_code_links":0,"syntology":null},{"paper":null,"slug":"hitting-the-high-dimensional-notes-an-ode-for","title":"Hitting the High-Dimensional Notes: An ODE for SGD learning dynamics on GLMs and multi-index models","date":"2023-08-17","arxiv_id":"2308.08977","n_code_links":0,"syntology":null},{"paper":null,"slug":"max-affine-regression-via-first-order-methods","title":"Max-affine regression via first-order methods","date":"2023-08-15","arxiv_id":"2308.08070","n_code_links":0,"syntology":null},{"paper":null,"slug":"law-of-balance-and-stationary-distribution-of","title":"Law of Balance and Stationary Distribution of Stochastic Gradient Descent","date":"2023-08-13","arxiv_id":"2308.06671","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-the-robustness-difference","slug":"understanding-the-robustness-difference","title":"Understanding the robustness difference between stochastic gradient descent and adaptive gradient methods","date":"2023-08-13","arxiv_id":"2308.06703","n_code_links":1,"syntology":{"ran":11,"of":13,"n_ran_checked":9,"n_instrument":2,"unverified":2,"pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["averyma/opt-robust"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"adaptive-sgd-with-polyak-stepsize-and-line","title":"Adaptive SGD with Polyak stepsize and Line-search: Robust Convergence and Variance Reduction","date":"2023-08-11","arxiv_id":"2308.06058","n_code_links":0,"syntology":null},{"paper":null,"slug":"real-time-progressive-learning-mutually","title":"Real-Time Progressive Learning: Accumulate Knowledge from Control with Neural-Network-Based Selective Memory","date":"2023-08-08","arxiv_id":"2308.04223","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-effect-of-sgd-batch-size-on-autoencoder","title":"The Effect of SGD Batch Size on Autoencoder Learning: Sparsity, Sharpness, and Feature Learning","date":"2023-08-06","arxiv_id":"2308.03215","n_code_links":0,"syntology":null},{"paper":null,"slug":"eva-a-general-vectorized-approximation","title":"Eva: A General Vectorized Approximation Framework for Second-order Optimization","date":"2023-08-04","arxiv_id":"2308.02123","n_code_links":0,"syntology":null},{"paper":"/paper/get-the-best-of-both-worlds-improving","slug":"get-the-best-of-both-worlds-improving","title":"Get the Best of Both Worlds: Improving Accuracy and Transferability by Grassmann Class Representation","date":"2023-08-03","arxiv_id":"2308.01547","n_code_links":1,"syntology":null},{"paper":null,"slug":"online-covariance-estimation-for-stochastic","title":"Online covariance estimation for stochastic gradient descent under Markovian sampling","date":"2023-08-03","arxiv_id":"2308.01481","n_code_links":0,"syntology":null},{"paper":null,"slug":"toward-quantum-machine-translation-of","title":"Toward Quantum Machine Translation of Syntactically Distinct Languages","date":"2023-07-31","arxiv_id":"2307.16576","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-marginal-value-of-momentum-for-small","title":"The Marginal Value of Momentum for Small Learning Rate SGD","date":"2023-07-27","arxiv_id":"2307.15196","n_code_links":0,"syntology":null},{"paper":null,"slug":"function-value-learning-adaptive-learning","title":"Function Value Learning: Adaptive Learning Rates Based on the Polyak Stepsize and Function Splitting in ERM","date":"2023-07-26","arxiv_id":"2307.14528","n_code_links":0,"syntology":null},{"paper":null,"slug":"relationship-between-batch-size-and-number-of","title":"Relationship between Batch Size and Number of Steps Needed for Nonconvex Optimization of Stochastic Gradient Descent using Armijo Line Search","date":"2023-07-25","arxiv_id":"2307.13831","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-of-sgd-for-training-neural","title":"Convergence of SGD for Training Neural Networks with Sliced Wasserstein Losses","date":"2023-07-21","arxiv_id":"2307.11714","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-guarantees-for-stochastic","title":"Stochastic Subgradient Methods with Guaranteed Global Stability in Nonsmooth Nonconvex Optimization","date":"2023-07-19","arxiv_id":"2307.10053","n_code_links":0,"syntology":null},{"paper":null,"slug":"weighted-averaged-stochastic-gradient-descent","title":"Weighted Averaged Stochastic Gradient Descent: Asymptotic Normality and Optimality","date":"2023-07-13","arxiv_id":"2307.06915","n_code_links":0,"syntology":null},{"paper":"/paper/mini-batch-optimization-of-contrastive-loss","slug":"mini-batch-optimization-of-contrastive-loss","title":"Mini-Batch Optimization of Contrastive Loss","date":"2023-07-12","arxiv_id":"2307.05906","n_code_links":1,"syntology":null},{"paper":null,"slug":"fedyolo-augmenting-federated-learning-with","title":"FedYolo: Augmenting Federated Learning with Pretrained Transformers","date":"2023-07-10","arxiv_id":"2307.04905","n_code_links":0,"syntology":null},{"paper":"/paper/bidirectional-looking-with-a-novel-double","slug":"bidirectional-looking-with-a-novel-double","title":"Bidirectional Looking with A Novel Double Exponential Moving Average to Adaptive and Non-adaptive Momentum Optimizers","date":"2023-07-02","arxiv_id":"2307.00631","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["chernyn/admeta-optimizer"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"federated-ensemble-yolov5-a-better","title":"Federated Ensemble YOLOv5 -- A Better Generalized Object Detection Algorithm","date":"2023-06-30","arxiv_id":"2306.17829","n_code_links":0,"syntology":null},{"paper":null,"slug":"nonconvex-stochastic-bregman-proximal","title":"Nonconvex Stochastic Bregman Proximal Gradient Method with Application to Deep Learning","date":"2023-06-26","arxiv_id":"2306.14522","n_code_links":0,"syntology":null},{"paper":null,"slug":"empirical-risk-minimization-with-shuffled-sgd","title":"Empirical Risk Minimization with Shuffled SGD: A Primal-Dual Perspective and Improved Bounds","date":"2023-06-21","arxiv_id":"2306.12498","n_code_links":0,"syntology":null},{"paper":"/paper/inrank-incremental-low-rank-learning","slug":"inrank-incremental-low-rank-learning","title":"InRank: Incremental Low-Rank Learning","date":"2023-06-20","arxiv_id":"2306.11250","n_code_links":1,"syntology":null},{"paper":"/paper/adaptive-federated-learning-with-auto-tuned","slug":"adaptive-federated-learning-with-auto-tuned","title":"Adaptive Federated Learning with Auto-Tuned Clients","date":"2023-06-19","arxiv_id":"2306.11201","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-strategies-in-non-convex","title":"Adaptive Strategies in Non-convex Optimization","date":"2023-06-17","arxiv_id":"2306.10278","n_code_links":0,"syntology":null},{"paper":"/paper/gradient-is-all-you-need","slug":"gradient-is-all-you-need","title":"Gradient is All You Need?","date":"2023-06-16","arxiv_id":"2306.09778","n_code_links":2,"syntology":null},{"paper":"/paper/evaluation-and-optimization-of-gradient","slug":"evaluation-and-optimization-of-gradient","title":"Evaluation and Optimization of Gradient Compression for Distributed Deep Learning","date":"2023-06-15","arxiv_id":"2306.08881","n_code_links":1,"syntology":null},{"paper":null,"slug":"span-selective-linear-attention-transformers","title":"Span-Selective Linear Attention Transformers for Effective and Robust Schema-Guided Dialogue State Tracking","date":"2023-06-15","arxiv_id":"2306.09340","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-re-weighted-gradient-descent-via","title":"Stochastic Re-weighted Gradient Descent via Distributionally Robust Optimization","date":"2023-06-15","arxiv_id":"2306.09222","n_code_links":0,"syntology":null},{"paper":null,"slug":"when-and-why-momentum-accelerates-sgd-an","title":"When and Why Momentum Accelerates SGD:An Empirical Study","date":"2023-06-15","arxiv_id":"2306.09000","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-implicit-bias-the-insignificance-of","title":"Beyond Implicit Bias: The Insignificance of SGD Noise in Online Learning","date":"2023-06-14","arxiv_id":"2306.08590","n_code_links":0,"syntology":null},{"paper":null,"slug":"kalman-filter-for-online-classification-of","title":"Kalman Filter for Online Classification of Non-Stationary Data","date":"2023-06-14","arxiv_id":"2306.08448","n_code_links":0,"syntology":null},{"paper":"/paper/noise-stability-optimization-for-flat-minima","slug":"noise-stability-optimization-for-flat-minima","title":"Noise Stability Optimization for Finding Flat Minima: A Hessian-based Regularization Approach","date":"2023-06-14","arxiv_id":"2306.08553","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":13,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["virtuosoresearch/noise-stability-optimization"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exact-mean-square-linear-stability-analysis","title":"Exact Mean Square Linear Stability Analysis for SGD","date":"2023-06-13","arxiv_id":"2306.07850","n_code_links":0,"syntology":null},{"paper":"/paper/implicit-compressibility-of-overparametrized","slug":"implicit-compressibility-of-overparametrized","title":"Implicit Compressibility of Overparametrized Neural Networks Trained with Heavy-Tailed SGD","date":"2023-06-13","arxiv_id":"2306.08125","n_code_links":1,"syntology":null},{"paper":null,"slug":"convergence-of-mean-field-langevin-dynamics","title":"Convergence of mean-field Langevin dynamics: Time and space discretization, stochastic gradient, and variance reduction","date":"2023-06-12","arxiv_id":"2306.07221","n_code_links":0,"syntology":null},{"paper":"/paper/fast-diffusion-model","slug":"fast-diffusion-model","title":"Fast Diffusion Model","date":"2023-06-12","arxiv_id":"2306.06991","n_code_links":1,"syntology":{"ran":14,"of":17,"n_ran_checked":8,"n_instrument":6,"unverified":3,"pointer_only":9,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 3 honoured, 0 violated, 5 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","official":{"repos":["sail-sg/fdm"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/gaussian-membership-inference-privacy-1","slug":"gaussian-membership-inference-privacy-1","title":"Gaussian Membership Inference Privacy","date":"2023-06-12","arxiv_id":"2306.07273","n_code_links":1,"syntology":null},{"paper":null,"slug":"correlated-noise-in-epoch-based-stochastic","title":"Correlated Noise in Epoch-Based Stochastic Gradient Descent: Implications for Weight Variances","date":"2023-06-08","arxiv_id":"2306.05300","n_code_links":0,"syntology":null}],"record_sha256":"18fb26c37db8a023cfa830a5c089129afb99a50acb448443638ce719ca4d5360","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}