{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/4","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":4,"pages_in_order":21,"rows_per_page":100,"rows":[301,400],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/3","next":"/method/sgd/papers/5","papers":[{"paper":null,"slug":"improving-implicit-regularization-of-sgd-with","title":"Improving Implicit Regularization of SGD with Preconditioning for Least Square Problems","date":"2024-03-13","arxiv_id":"2403.08585","n_code_links":0,"syntology":null},{"paper":"/paper/do-deep-neural-network-solutions-form-a-star","slug":"do-deep-neural-network-solutions-form-a-star","title":"Do Deep Neural Network Solutions Form a Star Domain?","date":"2024-03-12","arxiv_id":"2403.07968","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["aktsonthalia/starlight"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"efficient-language-model-architectures-for","title":"Efficient Language Model Architectures for Differentially Private Federated Learning","date":"2024-03-12","arxiv_id":"2403.08100","n_code_links":0,"syntology":null},{"paper":"/paper/sgd-with-partial-hessian-for-deep-neural","slug":"sgd-with-partial-hessian-for-deep-neural","title":"SGD with Partial Hessian for Deep Neural Networks Optimization","date":"2024-03-05","arxiv_id":"2403.02681","n_code_links":1,"syntology":null},{"paper":null,"slug":"shuffling-momentum-gradient-algorithm-for","title":"Shuffling Momentum Gradient Algorithm for Convex Optimization","date":"2024-03-05","arxiv_id":"2403.03180","n_code_links":0,"syntology":null},{"paper":null,"slug":"sofim-stochastic-optimization-using","title":"SOFIM: Stochastic Optimization Using Regularized Fisher Information Matrix","date":"2024-03-05","arxiv_id":"2403.02833","n_code_links":0,"syntology":null},{"paper":null,"slug":"differential-privacy-of-noisy-s-gd-under","title":"Privacy of SGD under Gaussian or Heavy-Tailed Noise: Guarantees without Gradient Clipping","date":"2024-03-04","arxiv_id":"2403.02051","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-implicit-bias-of-heterogeneity-towards","title":"The Implicit Bias of Heterogeneity towards Invariance: A Study of Multi-Environment Matrix Sensing","date":"2024-03-03","arxiv_id":"2403.01420","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-single-model-views-for-deep-learning","title":"Beyond Single-Model Views for Deep Learning: Optimization versus Generalizability of Stochastic Optimization Algorithms","date":"2024-03-01","arxiv_id":"2403.00574","n_code_links":0,"syntology":null},{"paper":"/paper/why-transformers-need-adam-a-hessian","slug":"why-transformers-need-adam-a-hessian","title":"Why Transformers Need Adam: A Hessian Perspective","date":"2024-02-26","arxiv_id":"2402.16788","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zyushun/hessian-spectrum"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"effective-gradient-sample-size-via-variation","title":"Effective Gradient Sample Size via Variation Estimation for Accelerating Sharpness aware Minimization","date":"2024-02-24","arxiv_id":"2403.08821","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-memory-based-adaptive-optimization","title":"Dynamic Memory Based Adaptive Optimization","date":"2024-02-23","arxiv_id":"2402.15262","n_code_links":0,"syntology":null},{"paper":"/paper/iteration-and-stochastic-first-order-oracle","slug":"iteration-and-stochastic-first-order-oracle","title":"Iteration and Stochastic First-order Oracle Complexities of Stochastic Gradient Descent using Constant and Decaying Learning Rates","date":"2024-02-23","arxiv_id":"2402.15344","n_code_links":1,"syntology":null},{"paper":null,"slug":"sgd-with-clipping-is-secretly-estimating-the","title":"SGD with Clipping is Secretly Estimating the Median Gradient","date":"2024-02-20","arxiv_id":"2402.12828","n_code_links":0,"syntology":null},{"paper":null,"slug":"training-artificial-neural-networks-by-1","title":"Training Artificial Neural Networks by Coordinate Search Algorithm","date":"2024-02-20","arxiv_id":"2402.12646","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-skeleton-graph-decoding","title":"Adaptive Skeleton Graph Decoding","date":"2024-02-19","arxiv_id":"2402.12280","n_code_links":0,"syntology":null},{"paper":null,"slug":"communication-efficient-distributed-learning-6","title":"Communication-Efficient Distributed Learning with Local Immediate Error Compensation","date":"2024-02-19","arxiv_id":"2402.11857","n_code_links":0,"syntology":null},{"paper":"/paper/diagonalisation-sgd-fast-convergent-sgd-for","slug":"diagonalisation-sgd-fast-convergent-sgd-for","title":"Diagonalisation SGD: Fast & Convergent SGD for Non-Differentiable Models via Reparameterisation and Smoothing","date":"2024-02-19","arxiv_id":"2402.11752","n_code_links":1,"syntology":null},{"paper":"/paper/optex-expediting-first-order-optimization","slug":"optex-expediting-first-order-optimization","title":"OptEx: Expediting First-Order Optimization with Approximately Parallelized Iterations","date":"2024-02-18","arxiv_id":"2402.11427","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":2,"n_instrument":5,"unverified":0,"pointer_only":7,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["youyve/OptEx"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/revisiting-zeroth-order-optimization-for","slug":"revisiting-zeroth-order-optimization-for","title":"Revisiting Zeroth-Order Optimization for Memory-Efficient LLM Fine-Tuning: A Benchmark","date":"2024-02-18","arxiv_id":"2402.11592","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":3,"n_instrument":6,"unverified":4,"pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","official":{"repos":["zo-bench/zo-llm"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"implicit-bias-in-noisy-sgd-with-applications","title":"Implicit Bias in Noisy-SGD: With Applications to Differentially Private Training","date":"2024-02-13","arxiv_id":"2402.08344","n_code_links":0,"syntology":null},{"paper":null,"slug":"differentially-private-zeroth-order-methods","title":"Differentially Private Zeroth-Order Methods for Scalable Large Language Model Finetuning","date":"2024-02-12","arxiv_id":"2402.07818","n_code_links":0,"syntology":null},{"paper":null,"slug":"tuning-free-stochastic-optimization","title":"Tuning-Free Stochastic Optimization","date":"2024-02-12","arxiv_id":"2402.07793","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-implicit-bias-of-gradient-noise-a","title":"Parameter Symmetry and Noise Equilibrium of Stochastic Gradient Descent","date":"2024-02-11","arxiv_id":"2402.07193","n_code_links":0,"syntology":null},{"paper":null,"slug":"should-i-try-multiple-optimizers-when-fine","title":"Should I try multiple optimizers when fine-tuning pre-trained Transformers for NLP tasks? Should I tune their hyperparameters?","date":"2024-02-10","arxiv_id":"2402.06948","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-learning-complexity-for-downstream","title":"Exploring Learning Complexity for Efficient Downstream Dataset Pruning","date":"2024-02-08","arxiv_id":"2402.05356","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-convergence-of-zeroth-order-federated","slug":"on-the-convergence-of-zeroth-order-federated","title":"On the Convergence of Zeroth-Order Federated Tuning for Large Language Models","date":"2024-02-08","arxiv_id":"2402.05926","n_code_links":1,"syntology":null},{"paper":null,"slug":"adabatchgrad-combining-adaptive-batch-size","title":"AdaBatchGrad: Combining Adaptive Batch Size and Adaptive Step Size","date":"2024-02-07","arxiv_id":"2402.05264","n_code_links":0,"syntology":null},{"paper":"/paper/curvature-informed-sgd-via-general-purpose","slug":"curvature-informed-sgd-via-general-purpose","title":"Curvature-Informed SGD via General Purpose Lie-Group Preconditioners","date":"2024-02-07","arxiv_id":"2402.04553","n_code_links":1,"syntology":null},{"paper":null,"slug":"gradient-descent-induces-alignment-between","title":"Feature learning as alignment: a structural property of gradient descent in non-linear neural networks","date":"2024-02-07","arxiv_id":"2402.05271","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-operators-with-stochastic-gradient","title":"Learning Operators with Stochastic Gradient Descent in General Hilbert Spaces","date":"2024-02-07","arxiv_id":"2402.04691","n_code_links":0,"syntology":null},{"paper":null,"slug":"non-convergence-to-global-minimizers-for-adam","title":"Non-convergence to global minimizers for Adam and stochastic gradient descent optimization and constructions of local minimizers in the training of artificial neural networks","date":"2024-02-07","arxiv_id":"2402.05155","n_code_links":0,"syntology":null},{"paper":null,"slug":"shadowheart-sgd-distributed-asynchronous-sgd","title":"Shadowheart SGD: Distributed Asynchronous SGD with Optimal Time Complexity Under Arbitrary Computation and Communication Heterogeneity","date":"2024-02-07","arxiv_id":"2402.04785","n_code_links":0,"syntology":null},{"paper":"/paper/can-we-remove-the-square-root-in-adaptive","slug":"can-we-remove-the-square-root-in-adaptive","title":"Can We Remove the Square-Root in Adaptive Gradient Methods? A Second-Order Perspective","date":"2024-02-05","arxiv_id":"2402.03496","n_code_links":2,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["f-dangel/sirfshampoo","yorkerlin/remove-the-square-root"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/non-asymptotic-analysis-of-biased-adaptive","slug":"non-asymptotic-analysis-of-biased-adaptive","title":"Non-asymptotic Analysis of Biased Adaptive Stochastic Approximation","date":"2024-02-05","arxiv_id":"2402.02857","n_code_links":1,"syntology":null},{"paper":"/paper/rethink-model-re-basin-and-the-linear-mode","slug":"rethink-model-re-basin-and-the-linear-mode","title":"Vanishing Feature: Diagnosing Model Merging and Beyond","date":"2024-02-05","arxiv_id":"2402.05966","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["xingyuqu/rethink-re-basin","xingyuqu/vf"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/riemannian-preconditioned-lora-for-fine","slug":"riemannian-preconditioned-lora-for-fine","title":"Riemannian Preconditioned LoRA for Fine-Tuning Foundation Models","date":"2024-02-04","arxiv_id":"2402.02347","n_code_links":1,"syntology":{"ran":15,"of":21,"n_ran_checked":11,"n_instrument":4,"unverified":6,"pointer_only":3,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","official":{"repos":["pilancilab/Riemannian_Preconditioned_LoRA"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"role-of-momentum-in-smoothing-objective","title":"Momentum Does Not Reduce Stochastic Noise in Stochastic Gradient Descent","date":"2024-02-04","arxiv_id":"2402.02325","n_code_links":0,"syntology":null},{"paper":null,"slug":"emergence-of-heavy-tails-in-homogenized","title":"Emergence of heavy tails in homogenized stochastic gradient descent","date":"2024-02-02","arxiv_id":"2402.01382","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-stochastic-gradient-descent-a","title":"Enhancing Stochastic Gradient Descent: A Unified Framework and Novel Acceleration Methods for Faster Convergence","date":"2024-02-02","arxiv_id":"2402.01515","n_code_links":0,"syntology":null},{"paper":null,"slug":"truncated-non-uniform-quantization-for","title":"Truncated Non-Uniform Quantization for Distributed SGD","date":"2024-02-02","arxiv_id":"2402.01160","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-adam-optimizer-via-online","title":"Understanding Adam Optimizer via Online Learning of Updates: Adam is FTRL in Disguise","date":"2024-02-02","arxiv_id":"2402.01567","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-spectral-bias-and-robustness-for","title":"Comparing Spectral Bias and Robustness For Two-Layer Neural Networks: SGD vs Adaptive Random Fourier Features","date":"2024-02-01","arxiv_id":"2402.00332","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-o-frac-sqrt-d-t-1-4-convergence-rate","title":"On the $O(\\frac{\\sqrt{d}}{T^{1/4}})$ Convergence Rate of RMSProp and Its Momentum Extension Measured by $\\ell_1$ Norm","date":"2024-02-01","arxiv_id":"2402.00389","n_code_links":0,"syntology":null},{"paper":"/paper/hift-a-hierarchical-full-parameter-fine","slug":"hift-a-hierarchical-full-parameter-fine","title":"HiFT: A Hierarchical Full Parameter Fine-Tuning Strategy","date":"2024-01-26","arxiv_id":"2401.15207","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["misonsky/HiFT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-principled-local-optimization-methods-for","title":"On Principled Local Optimization Methods for Federated Learning","date":"2024-01-24","arxiv_id":"2401.13216","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-precise-characterization-of-sgd-stability","title":"A Precise Characterization of SGD Stability Using Loss Surface Geometry","date":"2024-01-22","arxiv_id":"2401.12332","n_code_links":0,"syntology":null},{"paper":"/paper/momentum-sam-sharpness-aware-minimization","slug":"momentum-sam-sharpness-aware-minimization","title":"Momentum-SAM: Sharpness Aware Minimization without Computational Overhead","date":"2024-01-22","arxiv_id":"2401.12033","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["marlonbecker/msam"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-dimension-strikes-back-with-gradients","title":"The Dimension Strikes Back with Gradients: Generalization of Gradient Methods in Stochastic Convex Optimization","date":"2024-01-22","arxiv_id":"2401.12058","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-the-generalization-benefits-of","title":"Understanding the Generalization Benefits of Late Learning Rate Decay","date":"2024-01-21","arxiv_id":"2401.11600","n_code_links":0,"syntology":null},{"paper":"/paper/communication-efficient-and-provable","slug":"communication-efficient-and-provable","title":"Communication Efficient and Provable Federated Unlearning","date":"2024-01-19","arxiv_id":"2401.11018","n_code_links":1,"syntology":null},{"paper":"/paper/asynchronous-local-sgd-training-for-language","slug":"asynchronous-local-sgd-training-for-language","title":"Asynchronous Local-SGD Training for Language Modeling","date":"2024-01-17","arxiv_id":"2401.09135","n_code_links":1,"syntology":null},{"paper":null,"slug":"central-limit-theorem-for-two-timescale","title":"Central Limit Theorem for Two-Timescale Stochastic Approximation with Markovian Noise: Theory and Applications","date":"2024-01-17","arxiv_id":"2401.09339","n_code_links":0,"syntology":null},{"paper":"/paper/stabilizing-sharpness-aware-minimization","slug":"stabilizing-sharpness-aware-minimization","title":"Stabilizing Sharpness-aware Minimization Through A Simple Renormalization Strategy","date":"2024-01-14","arxiv_id":"2401.07250","n_code_links":2,"syntology":null},{"paper":null,"slug":"an-adrc-incorporated-stochastic-gradient","title":"An ADRC-Incorporated Stochastic Gradient Descent Algorithm for Latent Factor Analysis","date":"2024-01-13","arxiv_id":"2401.07012","n_code_links":0,"syntology":null},{"paper":"/paper/noise-adaptive-accelerated-stochastic-heavy","slug":"noise-adaptive-accelerated-stochastic-heavy","title":"(Accelerated) Noise-adaptive Stochastic Heavy-Ball Momentum","date":"2024-01-12","arxiv_id":"2401.06738","n_code_links":1,"syntology":null},{"paper":null,"slug":"correlated-quantization-for-faster-nonconvex","title":"Correlated Quantization for Faster Nonconvex Distributed Optimization","date":"2024-01-10","arxiv_id":"2401.05518","n_code_links":0,"syntology":null},{"paper":null,"slug":"aa-dladmm-an-accelerated-admm-based-framework","title":"AA-DLADMM: An Accelerated ADMM-based Framework for Training Deep Neural Networks","date":"2024-01-08","arxiv_id":"2401.03619","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-numerical-reliability-of-nonsmooth","slug":"on-the-numerical-reliability-of-nonsmooth","title":"On the numerical reliability of nonsmooth autodiff: a MaxPool case study","date":"2024-01-05","arxiv_id":"2401.02736","n_code_links":1,"syntology":null},{"paper":null,"slug":"ravnest-decentralized-asynchronous-training","title":"Ravnest: Decentralized Asynchronous Training on Heterogeneous Devices","date":"2024-01-03","arxiv_id":"2401.01728","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-tensor-inference","title":"Online Tensor Inference","date":"2023-12-28","arxiv_id":"2312.17111","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-trajectories-of-sgd-without","title":"On the Trajectories of SGD Without Replacement","date":"2023-12-26","arxiv_id":"2312.16143","n_code_links":0,"syntology":null},{"paper":"/paper/zo-adamu-optimizer-adapting-perturbation-by","slug":"zo-adamu-optimizer-adapting-perturbation-by","title":"ZO-AdaMU Optimizer: Adapting Perturbation by the Momentum and Uncertainty in Zeroth-order Optimization","date":"2023-12-23","arxiv_id":"2312.15184","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":4,"n_instrument":2,"unverified":4,"pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["mathisall/zo-adamu"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"accelerated-convergence-of-stochastic-heavy","title":"Accelerated Convergence of Stochastic Heavy Ball Method under Anisotropic Gradient Noise","date":"2023-12-22","arxiv_id":"2312.14567","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-convergence-of-loss-and-uncertainty","slug":"on-the-convergence-of-loss-and-uncertainty","title":"On the Convergence of Loss and Uncertainty-based Active Learning Algorithms","date":"2023-12-21","arxiv_id":"2312.13927","n_code_links":1,"syntology":null},{"paper":null,"slug":"parallel-trust-region-approaches-in-neural","title":"Parallel Trust-Region Approaches in Neural Network Training: Beyond Traditional Methods","date":"2023-12-21","arxiv_id":"2312.13677","n_code_links":0,"syntology":null},{"paper":null,"slug":"contractive-error-feedback-for-gradient-1","title":"Contractive error feedback for gradient compression","date":"2023-12-13","arxiv_id":"2312.08538","n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-the-last-iterate-convergence-of","title":"Revisiting the Last-Iterate Convergence of Stochastic Gradient Methods","date":"2023-12-13","arxiv_id":"2312.08531","n_code_links":0,"syntology":null},{"paper":"/paper/tod-flow-modeling-the-structure-of-task","slug":"tod-flow-modeling-the-structure-of-task","title":"TOD-Flow: Modeling the Structure of Task-Oriented Dialogues","date":"2023-12-07","arxiv_id":"2312.04668","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["srsohn/tod-flow"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"convergence-rates-for-stochastic-1","title":"Convergence Rates for Stochastic Approximation: Biased Noise with Unbounded Variance, and Applications","date":"2023-12-05","arxiv_id":"2312.02828","n_code_links":0,"syntology":null},{"paper":"/paper/agd-an-auto-switchable-optimizer-using-1","slug":"agd-an-auto-switchable-optimizer-using-1","title":"AGD: an Auto-switchable Optimizer using Stepwise Gradient Difference for Preconditioning Matrix","date":"2023-12-04","arxiv_id":"2312.01658","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["intelligent-machine-learning/atorch","intelligent-machine-learning/dlrover"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"paper":null,"slug":"unlocking-optimal-batch-size-schedules-using","title":"Unlocking optimal batch size schedules using continuous-time control and perturbation theory","date":"2023-12-04","arxiv_id":"2312.01898","n_code_links":0,"syntology":null},{"paper":"/paper/can-we-learn-communication-efficient","slug":"can-we-learn-communication-efficient","title":"Can We Learn Communication-Efficient Optimizers?","date":"2023-12-02","arxiv_id":"2312.02204","n_code_links":1,"syntology":null},{"paper":"/paper/temperature-balancing-layer-wise-weight-1","slug":"temperature-balancing-layer-wise-weight-1","title":"Temperature Balancing, Layer-wise Weight Analysis, and Neural Network Training","date":"2023-12-01","arxiv_id":"2312.00359","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yefanzhou/tempbalance"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"steering-deep-feature-learning-with-backward","title":"The Feature Speed Formula: a flexible approach to scale hyper-parameters of deep neural networks","date":"2023-11-30","arxiv_id":"2311.18718","n_code_links":0,"syntology":null},{"paper":"/paper/the-effects-of-overparameterization-on","slug":"the-effects-of-overparameterization-on","title":"Critical Influence of Overparameterization on Sharpness-aware Minimization","date":"2023-11-29","arxiv_id":"2311.17539","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["log-postech/sam-overparam"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/in-search-of-a-data-transformation-that","slug":"in-search-of-a-data-transformation-that","title":"In Search of a Data Transformation That Accelerates Neural Field Training","date":"2023-11-28","arxiv_id":"2311.17094","n_code_links":1,"syntology":null},{"paper":"/paper/mast-model-agnostic-sparsified-training","slug":"mast-model-agnostic-sparsified-training","title":"MAST: Model-Agnostic Sparsified Training","date":"2023-11-27","arxiv_id":"2311.16086","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["konstmish/opt_methods"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"scheduling-and-communication-schemes-for","title":"Scheduling and Communication Schemes for Decentralized Federated Learning","date":"2023-11-27","arxiv_id":"2311.16021","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-gradient-estimation-via-adaptive","title":"Online Importance Sampling for Stochastic Gradient Optimization","date":"2023-11-24","arxiv_id":"2311.14468","n_code_links":0,"syntology":null},{"paper":null,"slug":"risk-bounds-of-accelerated-sgd-for","title":"Risk Bounds of Accelerated SGD for Overparameterized Linear Regression","date":"2023-11-23","arxiv_id":"2311.14222","n_code_links":0,"syntology":null},{"paper":null,"slug":"weight-fluctuations-in-deep-linear-neural","title":"Weight fluctuations in (deep) linear neural networks and a derivation of the inverse-variance flatness relation","date":"2023-11-23","arxiv_id":"2311.14120","n_code_links":0,"syntology":null},{"paper":null,"slug":"sample-as-you-infer-predictive-coding-with","title":"Sample as You Infer: Predictive Coding With Langevin Dynamics","date":"2023-11-22","arxiv_id":"2311.13664","n_code_links":0,"syntology":null},{"paper":"/paper/expm-nf-differentially-private-machine","slug":"expm-nf-differentially-private-machine","title":"Are Normalizing Flows the Key to Unlocking the Exponential Mechanism?","date":"2023-11-15","arxiv_id":"2311.09200","n_code_links":1,"syntology":null},{"paper":null,"slug":"dagc-data-volume-aware-adaptive","title":"Data-Aware Gradient Compression for FL in Communication-Constrained Mobile Computing","date":"2023-11-13","arxiv_id":"2311.07324","n_code_links":0,"syntology":null},{"paper":null,"slug":"malcom-psgd-inexact-proximal-stochastic","title":"Compressed and Sparse Models for Non-Convex Decentralized Learning","date":"2023-11-09","arxiv_id":"2311.05760","n_code_links":0,"syntology":null},{"paper":null,"slug":"outliers-with-opposing-signals-have-an","title":"Outliers with Opposing Signals Have an Outsized Effect on Neural Network Optimization","date":"2023-11-07","arxiv_id":"2311.04163","n_code_links":0,"syntology":null},{"paper":"/paper/signal-processing-meets-sgd-from-momentum-to","slug":"signal-processing-meets-sgd-from-momentum-to","title":"Signal Processing Meets SGD: From Momentum to Filter","date":"2023-11-06","arxiv_id":"2311.02818","n_code_links":1,"syntology":null},{"paper":null,"slug":"asynchronous-sgd-on-graphs-a-unified","title":"Asynchronous SGD on Graphs: a Unified Framework for Asynchronous Decentralized and Federated Optimization","date":"2023-11-01","arxiv_id":"2311.00465","n_code_links":0,"syntology":null},{"paper":null,"slug":"asgrad-a-sharp-unified-analysis-of","title":"AsGrad: A Sharp Unified Analysis of Asynchronous-SGD Algorithms","date":"2023-10-31","arxiv_id":"2310.20452","n_code_links":0,"syntology":null},{"paper":"/paper/information-theoretic-trust-regions-for","slug":"information-theoretic-trust-regions-for","title":"Information-Theoretic Trust Regions for Stochastic Gradient-Based Optimization","date":"2023-10-31","arxiv_id":"2310.20574","n_code_links":1,"syntology":null},{"paper":null,"slug":"escaping-saddle-points-in-heterogeneous","title":"Escaping Saddle Points in Heterogeneous Federated Learning via Distributed SGD with Communication Compression","date":"2023-10-29","arxiv_id":"2310.19059","n_code_links":0,"syntology":null},{"paper":null,"slug":"high-probability-convergence-bounds-for-1","title":"High-probability Convergence Bounds for Nonlinear Stochastic Gradient Descent Under Heavy-tailed Noise","date":"2023-10-28","arxiv_id":"2310.18784","n_code_links":0,"syntology":null},{"paper":null,"slug":"linear-mode-connectivity-in-sparse-neural","title":"Linear Mode Connectivity in Sparse Neural Networks","date":"2023-10-28","arxiv_id":"2310.18769","n_code_links":0,"syntology":null},{"paper":null,"slug":"benign-oscillation-of-stochastic-gradient","title":"Benign Oscillation of Stochastic Gradient Descent with Large Learning Rates","date":"2023-10-26","arxiv_id":"2310.17074","n_code_links":0,"syntology":null},{"paper":"/paper/grokking-beyond-neural-networks-an-empirical","slug":"grokking-beyond-neural-networks-an-empirical","title":"Grokking Beyond Neural Networks: An Empirical Exploration with Model Complexity","date":"2023-10-26","arxiv_id":"2310.17247","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-model-for-multi-attack-classification-to","title":"A model for multi-attack classification to improve intrusion detection performance using deep learning approaches","date":"2023-10-25","arxiv_id":"2310.16380","n_code_links":0,"syntology":null},{"paper":null,"slug":"probabilistic-integral-circuits","title":"Probabilistic Integral Circuits","date":"2023-10-25","arxiv_id":"2310.16986","n_code_links":0,"syntology":null},{"paper":"/paper/learning-from-free-text-human-feedback","slug":"learning-from-free-text-human-feedback","title":"Learning From Free-Text Human Feedback -- Collect New Datasets Or Extend Existing Ones?","date":"2023-10-24","arxiv_id":"2310.15758","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ukplab/emnlp2023-learning-from-free-text-human-feedback"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/adam-through-a-second-order-lens","slug":"adam-through-a-second-order-lens","title":"Studying K-FAC Heuristics by Viewing Adam through a Second-Order Lens","date":"2023-10-23","arxiv_id":"2310.14963","n_code_links":1,"syntology":{"ran":7,"of":16,"n_ran_checked":7,"n_instrument":0,"unverified":9,"pointer_only":16,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","official":{"repos":["rmclarke/adamthroughasecondorderlens"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":9,"ran_from_kinds":["official"]}}}],"record_sha256":"21496c15763adc74979024b3f64900e51b92dd6db92f17812d58cc587045123b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}