{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/6","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":6,"pages_in_order":21,"rows_per_page":100,"rows":[501,600],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/5","next":"/method/sgd/papers/7","papers":[{"paper":"/paper/catapults-in-sgd-spikes-in-the-training-loss","slug":"catapults-in-sgd-spikes-in-the-training-loss","title":"Catapults in SGD: spikes in the training loss and their impact on generalization through feature learning","date":"2023-06-07","arxiv_id":"2306.04815","n_code_links":1,"syntology":null},{"paper":"/paper/prompter-zero-shot-adaptive-prefixes-for","slug":"prompter-zero-shot-adaptive-prefixes-for","title":"Prompter: Zero-shot Adaptive Prefixes for Dialogue State Tracking Domain Adaptation","date":"2023-06-07","arxiv_id":"2306.04724","n_code_links":1,"syntology":null},{"paper":"/paper/stochastic-collapse-how-gradient-noise-1","slug":"stochastic-collapse-how-gradient-noise-1","title":"Stochastic Collapse: How Gradient Noise Attracts SGD Dynamics Towards Simpler Subnetworks","date":"2023-06-07","arxiv_id":"2306.04251","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ccffccffcc/stochastic_collapse"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/machine-learning-in-and-out-of-equilibrium","slug":"machine-learning-in-and-out-of-equilibrium","title":"Machine learning in and out of equilibrium","date":"2023-06-06","arxiv_id":"2306.03521","n_code_links":1,"syntology":null},{"paper":"/paper/decentralized-sgd-and-average-direction-sam","slug":"decentralized-sgd-and-average-direction-sam","title":"Decentralized SGD and Average-direction SAM are Asymptotically Equivalent","date":"2023-06-05","arxiv_id":"2306.02913","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":6,"n_instrument":2,"unverified":2,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["raiden-zhu/icml-2023-dsgd-and-sam"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improved-stability-and-generalization","title":"Improved Stability and Generalization Guarantees of the Decentralized SGD Algorithm","date":"2023-06-05","arxiv_id":"2306.02939","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-bootstrap-inference-with-nonconvex","title":"Online Bootstrap Inference with Nonconvex Stochastic Gradient Descent Estimator","date":"2023-06-03","arxiv_id":"2306.02205","n_code_links":0,"syntology":null},{"paper":"/paper/towards-sustainable-learning-coresets-for","slug":"towards-sustainable-learning-coresets-for","title":"Towards Sustainable Learning: Coresets for Data-efficient Deep Learning","date":"2023-06-02","arxiv_id":"2306.01244","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bigml-cs-ucla/crest"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-energy-conserving-descent-for","slug":"improving-energy-conserving-descent-for","title":"Improving Energy Conserving Descent for Machine Learning: Theory and Practice","date":"2023-06-01","arxiv_id":"2306.00352","n_code_links":1,"syntology":null},{"paper":"/paper/a-bayesian-approach-to-analysing-training-1","slug":"a-bayesian-approach-to-analysing-training-1","title":"A Bayesian Approach To Analysing Training Data Attribution In Deep Learning","date":"2023-05-31","arxiv_id":"2305.19765","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["elisanguyen/bayesian-tda"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-impact-of-positional-encoding-on-length-1","slug":"the-impact-of-positional-encoding-on-length-1","title":"The Impact of Positional Encoding on Length Generalization in Transformers","date":"2023-05-31","arxiv_id":"2305.19466","n_code_links":2,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mcgill-nlp/length-generalization"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"toward-understanding-why-adam-converges","title":"Toward Understanding Why Adam Converges Faster Than SGD for Transformers","date":"2023-05-31","arxiv_id":"2306.00204","n_code_links":0,"syntology":null},{"paper":null,"slug":"bisls-sps-auto-tune-step-sizes-for-stable-bi","title":"BiSLS/SPS: Auto-tune Step Sizes for Stable Bi-level Optimization","date":"2023-05-30","arxiv_id":"2305.18666","n_code_links":0,"syntology":null},{"paper":null,"slug":"shuffle-sgd-is-always-better-than-sgd","title":"On Convergence of Incremental Gradient for Non-Convex Smooth Functions","date":"2023-05-30","arxiv_id":"2305.19259","n_code_links":0,"syntology":null},{"paper":"/paper/a-rainbow-in-deep-network-black-boxes","slug":"a-rainbow-in-deep-network-black-boxes","title":"A Rainbow in Deep Network Black Boxes","date":"2023-05-29","arxiv_id":"2305.18512","n_code_links":1,"syntology":{"ran":13,"of":14,"n_ran_checked":11,"n_instrument":2,"unverified":1,"pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["florentinguth/rainbow"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"convergence-of-adagrad-for-non-convex","title":"Convergence of AdaGrad for Non-convex Objectives: Simple Proofs and Relaxed Assumptions","date":"2023-05-29","arxiv_id":"2305.18471","n_code_links":0,"syntology":null},{"paper":"/paper/escaping-mediocrity-how-two-layer-networks","slug":"escaping-mediocrity-how-two-layer-networks","title":"Escaping mediocrity: how two-layer networks learn hard generalized linear models with SGD","date":"2023-05-29","arxiv_id":"2305.18502","n_code_links":2,"syntology":null},{"paper":null,"slug":"acceleration-of-stochastic-gradient-descent","title":"Acceleration of stochastic gradient descent with momentum by averaging: finite-sample rates and asymptotic normality","date":"2023-05-28","arxiv_id":"2305.17665","n_code_links":0,"syntology":null},{"paper":"/paper/direction-oriented-multi-objective-learning","slug":"direction-oriented-multi-objective-learning","title":"Direction-oriented Multi-objective Learning: Simple and Provable Stochastic Algorithms","date":"2023-05-28","arxiv_id":"2305.18409","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":8,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ml-opt-lab/sdmgrad"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-implicit-regularization-of-dynamical","title":"The Implicit Regularization of Dynamical Stability in Stochastic Gradient Descent","date":"2023-05-27","arxiv_id":"2305.17490","n_code_links":0,"syntology":null},{"paper":"/paper/rotational-optimizers-simple-robust-dnn","slug":"rotational-optimizers-simple-robust-dnn","title":"Rotational Equilibrium: How Weight Decay Balances Learning Across Neural Networks","date":"2023-05-26","arxiv_id":"2305.17212","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["epfml/req","epfml/rotational-optimizers"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/schema-guided-user-satisfaction-modeling-for","slug":"schema-guided-user-satisfaction-modeling-for","title":"Schema-Guided User Satisfaction Modeling for Task-Oriented Dialogues","date":"2023-05-26","arxiv_id":"2305.16798","n_code_links":1,"syntology":null},{"paper":"/paper/xgrad-boosting-gradient-based-optimizers-with","slug":"xgrad-boosting-gradient-based-optimizers-with","title":"XGrad: Boosting Gradient-Based Optimizers With Weight Prediction","date":"2023-05-26","arxiv_id":"2305.18240","n_code_links":1,"syntology":null},{"paper":null,"slug":"adler-an-efficient-hessian-based-strategy-for","title":"ADLER -- An efficient Hessian-based strategy for adaptive learning rate","date":"2023-05-25","arxiv_id":"2305.16396","n_code_links":0,"syntology":null},{"paper":"/paper/exploiting-noise-as-a-resource-for","slug":"exploiting-noise-as-a-resource-for","title":"Exploiting Noise as a Resource for Computation and Learning in Spiking Neural Networks","date":"2023-05-25","arxiv_id":"2305.16044","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["genema/Noisy-Spiking-Neuron-Nets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"implicit-bias-of-sgd-in-l-2-regularized","title":"Implicit bias of SGD in $L_{2}$-regularized linear DNNs: One-way jumps from high to low rank","date":"2023-05-25","arxiv_id":"2305.16038","n_code_links":0,"syntology":null},{"paper":null,"slug":"incentivizing-honesty-among-competitors-in","title":"Incentivizing Honesty among Competitors in Collaborative Learning and Optimization","date":"2023-05-25","arxiv_id":"2305.16272","n_code_links":0,"syntology":null},{"paper":"/paper/neural-characteristic-activation-value","slug":"neural-characteristic-activation-value","title":"Neural Characteristic Activation Analysis and Geometric Parameterization for ReLU Networks","date":"2023-05-25","arxiv_id":"2305.15912","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":1,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["Wenlin-Chen/geometric-parameterization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"scan-and-snap-understanding-training-dynamics","title":"Scan and Snap: Understanding Training Dynamics and Token Composition in 1-layer Transformer","date":"2023-05-25","arxiv_id":"2305.16380","n_code_links":0,"syntology":null},{"paper":null,"slug":"2305-14859","title":"Utility-Probability Duality of Neural Networks","date":"2023-05-24","arxiv_id":"2305.14859","n_code_links":0,"syntology":null},{"paper":null,"slug":"2305-15013","title":"Local SGD Accelerates Convergence by Exploiting Second Order Information of the Loss Function","date":"2023-05-24","arxiv_id":"2305.15013","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-black-box-variational","title":"On the Convergence of Black-Box Variational Inference","date":"2023-05-24","arxiv_id":"2305.15349","n_code_links":0,"syntology":null},{"paper":null,"slug":"layer-wise-adaptive-step-sizes-for-stochastic","title":"Layer-wise Adaptive Step-Sizes for Stochastic First-Order Methods for Deep Learning","date":"2023-05-23","arxiv_id":"2305.13664","n_code_links":0,"syntology":null},{"paper":null,"slug":"two-sides-of-one-coin-the-limits-of-untuned","title":"Two Sides of One Coin: the Limits of Untuned SGD and the Power of Adaptive Methods","date":"2023-05-21","arxiv_id":"2305.12475","n_code_links":0,"syntology":null},{"paper":null,"slug":"evolutionary-algorithms-in-the-light-of-sgd","title":"Evolutionary Algorithms in the Light of SGD: Limit Equivalence, Minima Flatness, and Transfer Learning","date":"2023-05-20","arxiv_id":"2306.09991","n_code_links":0,"syntology":null},{"paper":"/paper/gravac-adaptive-compression-for-communication","slug":"gravac-adaptive-compression-for-communication","title":"GraVAC: Adaptive Compression for Communication-Efficient Distributed DL Training","date":"2023-05-20","arxiv_id":"2305.12201","n_code_links":1,"syntology":null},{"paper":null,"slug":"stability-and-generalization-of-ell-p","title":"Stability and Generalization of lp-Regularized Stochastic Learning for GCN","date":"2023-05-20","arxiv_id":"2305.12085","n_code_links":0,"syntology":null},{"paper":null,"slug":"uniform-in-time-wasserstein-stability-bounds","title":"Uniform-in-Time Wasserstein Stability Bounds for (Noisy) Stochastic Gradient Descent","date":"2023-05-20","arxiv_id":"2305.12056","n_code_links":0,"syntology":null},{"paper":null,"slug":"conditional-online-learning-for-keyword","title":"Conditional Online Learning for Keyword Spotting","date":"2023-05-19","arxiv_id":"2305.13332","n_code_links":0,"syntology":null},{"paper":null,"slug":"smoothing-the-landscape-boosts-the-signal-for","title":"Smoothing the Landscape Boosts the Signal for SGD: Optimal Sample Complexity for Learning Single Index Models","date":"2023-05-18","arxiv_id":"2305.10633","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-ratios-tracking-algorithm-for","title":"Stochastic Ratios Tracking Algorithm for Large Scale Machine Learning Problems","date":"2023-05-17","arxiv_id":"2305.09978","n_code_links":0,"syntology":null},{"paper":null,"slug":"faster-federated-learning-with-decaying-1","title":"Faster Federated Learning with Decaying Number of Local SGD Steps","date":"2023-05-16","arxiv_id":"2305.09628","n_code_links":0,"syntology":null},{"paper":"/paper/momo-momentum-models-for-adaptive-learning","slug":"momo-momentum-models-for-adaptive-learning","title":"MoMo: Momentum Models for Adaptive Learning Rates","date":"2023-05-12","arxiv_id":"2305.07583","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fabian-sp/MoMo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"online-learning-under-a-separable-stochastic","title":"Online Learning Under A Separable Stochastic Approximation Framework","date":"2023-05-12","arxiv_id":"2305.07484","n_code_links":0,"syntology":null},{"paper":"/paper/securing-distributed-sgd-against-gradient","slug":"securing-distributed-sgd-against-gradient","title":"Securing Distributed SGD against Gradient Leakage Threats","date":"2023-05-10","arxiv_id":"2305.06473","n_code_links":1,"syntology":null},{"paper":null,"slug":"fedhb-hierarchical-bayesian-federated","title":"FedHB: Hierarchical Bayesian Federated Learning","date":"2023-05-08","arxiv_id":"2305.04979","n_code_links":0,"syntology":null},{"paper":null,"slug":"semi-asynchronous-federated-edge-learning","title":"Semi-Asynchronous Federated Edge Learning Mechanism via Over-the-air Computation","date":"2023-05-06","arxiv_id":"2305.04066","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-momentum-incorporated-non-negative-latent","title":"A Momentum-Incorporated Non-Negative Latent Factorization of Tensors Model for Dynamic Network Representation","date":"2023-05-04","arxiv_id":"2305.02782","n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-gradient-clipping-stochastic-bias","title":"Revisiting Gradient Clipping: Stochastic bias and tight convergence guarantees","date":"2023-05-02","arxiv_id":"2305.01588","n_code_links":0,"syntology":null},{"paper":null,"slug":"fairness-uncertainty-quantification-how","title":"Fairness Uncertainty Quantification: How certain are you that the model is fair?","date":"2023-04-27","arxiv_id":"2304.13950","n_code_links":0,"syntology":null},{"paper":"/paper/noise-is-not-the-main-factor-behind-the-gap","slug":"noise-is-not-the-main-factor-behind-the-gap","title":"Noise Is Not the Main Factor Behind the Gap Between SGD and Adam on Transformers, but Sign Descent Might Be","date":"2023-04-27","arxiv_id":"2304.13960","n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchical-weight-averaging-for-deep-neural","title":"Hierarchical Weight Averaging for Deep Neural Networks","date":"2023-04-23","arxiv_id":"2304.11519","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-incomplete-tensor-tucker-decomposition","title":"An Incomplete Tensor Tucker decomposition based Traffic Speed Prediction Method","date":"2023-04-21","arxiv_id":"2304.10961","n_code_links":0,"syntology":null},{"paper":null,"slug":"loss-minimization-yields-multicalibration-for","title":"Loss Minimization Yields Multicalibration for Large Neural Networks","date":"2023-04-19","arxiv_id":"2304.09424","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-of-stochastic-gradient-descent-2","title":"Convergence of stochastic gradient descent under a local Lojasiewicz condition for deep neural networks","date":"2023-04-18","arxiv_id":"2304.09221","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-and-straggler-tolerant-distributed-sgd","title":"Fast and Straggler-Tolerant Distributed SGD with Reduced Computation Load","date":"2023-04-17","arxiv_id":"2304.08589","n_code_links":0,"syntology":null},{"paper":null,"slug":"landslide-susceptibility-prediction-modeling","title":"Landslide Susceptibility Prediction Modeling Based on Self-Screening Deep Learning Model","date":"2023-04-12","arxiv_id":"2304.06054","n_code_links":0,"syntology":null},{"paper":null,"slug":"high-dimensional-scaling-limits-and","title":"High-dimensional scaling limits and fluctuations of online least-squares SGD with smooth covariance","date":"2023-04-03","arxiv_id":"2304.00707","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-convergence-of-random-reshuffling-under","title":"Fast Convergence of Random Reshuffling under Over-Parameterization and the Polyak-Łojasiewicz Condition","date":"2023-04-02","arxiv_id":"2304.00459","n_code_links":0,"syntology":null},{"paper":null,"slug":"doubly-stochastic-models-learning-with","title":"Doubly Stochastic Models: Learning with Unbiased Label Noises and Inference Stability","date":"2023-04-01","arxiv_id":"2304.00320","n_code_links":0,"syntology":null},{"paper":null,"slug":"non-asymptotic-lower-bounds-for-training-data","title":"On the Query Complexity of Training Data Reconstruction in Private Learning","date":"2023-03-29","arxiv_id":"2303.16372","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-generalizable-end-to-end-task","slug":"zero-shot-generalizable-end-to-end-task","title":"Zero-Shot Generalizable End-to-End Task-Oriented Dialog System using Context Summarization and Domain Schema","date":"2023-03-28","arxiv_id":"2303.16252","n_code_links":1,"syntology":null},{"paper":null,"slug":"toward-open-domain-slot-filling-via-self","title":"Toward Open-domain Slot Filling via Self-supervised Co-training","date":"2023-03-24","arxiv_id":"2303.13801","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-probabilistic-stability-of-stochastic","title":"Type-II Saddles and Probabilistic Stability of Stochastic Gradient Descent","date":"2023-03-23","arxiv_id":"2303.13093","n_code_links":0,"syntology":null},{"paper":null,"slug":"mathcal-c-k-continuous-spline-approximation","title":"$\\mathcal{C}^k$-continuous Spline Approximation with TensorFlow Gradient Descent Optimizers","date":"2023-03-22","arxiv_id":"2303.12454","n_code_links":0,"syntology":null},{"paper":null,"slug":"lower-generalization-bounds-for-gd-and-sgd-in","title":"Lower Generalization Bounds for GD and SGD in Smooth Stochastic Convex Optimization","date":"2023-03-19","arxiv_id":"2303.10758","n_code_links":0,"syntology":null},{"paper":null,"slug":"provable-convergence-of-variational-monte","title":"Convergence Analysis of Stochastic Gradient Descent with MCMC Estimators","date":"2023-03-19","arxiv_id":"2303.10599","n_code_links":0,"syntology":null},{"paper":null,"slug":"practical-and-matching-gradient-variance","title":"Practical and Matching Gradient Variance Bounds for Black-Box Variational Bayesian Inference","date":"2023-03-18","arxiv_id":"2303.10472","n_code_links":0,"syntology":null},{"paper":"/paper/tighter-lower-bounds-for-shuffling-sgd-random","slug":"tighter-lower-bounds-for-shuffling-sgd-random","title":"Tighter Lower Bounds for Shuffling SGD: Random Permutations and Beyond","date":"2023-03-13","arxiv_id":"2303.07160","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"scavenger-a-cloud-service-for-optimizing-cost","title":"Scavenger: A Cloud Service for Optimizing Cost and Performance of ML Training","date":"2023-03-12","arxiv_id":"2303.06659","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-latent-factor-analysis-via-a-fuzzy-pid","title":"Fast Latent Factor Analysis via a Fuzzy PID-Incorporated Stochastic Gradient Descent Algorithm","date":"2023-03-07","arxiv_id":"2303.03941","n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-the-noise-model-of-stochastic","title":"Revisiting the Noise Model of Stochastic Gradient Descent","date":"2023-03-05","arxiv_id":"2303.02749","n_code_links":0,"syntology":null},{"paper":"/paper/gradient-norm-aware-minimization-seeks-first","slug":"gradient-norm-aware-minimization-seeks-first","title":"Gradient Norm Aware Minimization Seeks First-Order Flatness and Improves Generalization","date":"2023-03-03","arxiv_id":"2303.03108","n_code_links":1,"syntology":null},{"paper":null,"slug":"implicit-stochastic-gradient-descent-for","title":"Implicit Stochastic Gradient Descent for Training Physics-informed Neural Networks","date":"2023-03-03","arxiv_id":"2303.01767","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-high-dimensional-single-neuron-relu","title":"Finite-Sample Analysis of Learning High-Dimensional Single ReLU Neuron","date":"2023-03-03","arxiv_id":"2303.02255","n_code_links":0,"syntology":null},{"paper":"/paper/dropout-reduces-underfitting","slug":"dropout-reduces-underfitting","title":"Dropout Reduces Underfitting","date":"2023-03-02","arxiv_id":"2303.01500","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["facebookresearch/dropout"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"over-training-with-mixup-may-hurt","title":"Over-training with Mixup May Hurt Generalization","date":"2023-03-02","arxiv_id":"2303.01475","n_code_links":0,"syntology":null},{"paper":"/paper/variance-reduced-clipping-for-non-convex","slug":"variance-reduced-clipping-for-non-convex","title":"Variance-reduced Clipping for Non-convex Optimization","date":"2023-03-02","arxiv_id":"2303.00883","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["haochuan-mit/varaince-reduced-clipping-for-non-convex-optimization"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/why-and-when-does-local-sgd-generalize-better","slug":"why-and-when-does-local-sgd-generalize-better","title":"Why (and When) does Local SGD Generalize Better than SGD?","date":"2023-03-02","arxiv_id":"2303.01215","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hmgxr128/local-sgd"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"adasam-boosting-sharpness-aware-minimization","title":"AdaSAM: Boosting Sharpness-Aware Minimization with Adaptive Learning Rate and Momentum for Training Deep Neural Networks","date":"2023-03-01","arxiv_id":"2303.00565","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-efficient-tester-learner-for-halfspaces","title":"An Efficient Tester-Learner for Halfspaces","date":"2023-02-28","arxiv_id":"2302.14853","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-as-chita-neural-network-pruning-with","title":"Fast as CHITA: Neural Network Pruning with Combinatorial Optimization","date":"2023-02-28","arxiv_id":"2302.14623","n_code_links":0,"syntology":null},{"paper":null,"slug":"high-probability-convergence-of-stochastic","title":"High Probability Convergence of Stochastic Gradient Methods","date":"2023-02-28","arxiv_id":"2302.14843","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-under-markovian","title":"Stochastic Gradient Descent under Markovian Sampling Schemes","date":"2023-02-28","arxiv_id":"2302.14428","n_code_links":0,"syntology":null},{"paper":"/paper/a-unified-framework-for-soft-threshold","slug":"a-unified-framework-for-soft-threshold","title":"A Unified Framework for Soft Threshold Pruning","date":"2023-02-25","arxiv_id":"2302.13019","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yanqi-chen/lats"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-the-training-instability-of-shuffling-sgd","title":"On the Training Instability of Shuffling SGD with Batch Normalization","date":"2023-02-24","arxiv_id":"2302.12444","n_code_links":0,"syntology":null},{"paper":null,"slug":"statistical-inference-with-stochastic-1","title":"Statistical Inference with Stochastic Gradient Methods under $φ$-mixing Data","date":"2023-02-24","arxiv_id":"2302.12717","n_code_links":0,"syntology":null},{"paper":"/paper/random-teachers-are-good-teachers","slug":"random-teachers-are-good-teachers","title":"Random Teachers are Good Teachers","date":"2023-02-23","arxiv_id":"2302.12091","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-message-shuffled-privacy-in-federated","title":"Multi-Message Shuffled Privacy in Federated Learning","date":"2023-02-22","arxiv_id":"2302.11152","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-learning-on-neural-networks-leap","title":"SGD learning on neural networks: leap complexity and saddle-to-saddle dynamics","date":"2023-02-21","arxiv_id":"2302.11055","n_code_links":0,"syntology":null},{"paper":null,"slug":"high-dimensional-central-limit-theorems-for-1","title":"Statistical Inference for Linear Functionals of Online SGD in High-dimensional Linear Regression","date":"2023-02-20","arxiv_id":"2302.09727","n_code_links":0,"syntology":null},{"paper":null,"slug":"msam-micro-batch-averaged-sharpness-aware","title":"mSAM: Micro-Batch-Averaged Sharpness-Aware Minimization","date":"2023-02-19","arxiv_id":"2302.09693","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-generalization-error-of-stochastic-mirror","title":"The Generalization Error of Stochastic Mirror Descent on Over-Parametrized Linear Models","date":"2023-02-18","arxiv_id":"2302.09433","n_code_links":0,"syntology":null},{"paper":null,"slug":"s-gd-over-diagonal-linear-networks-implicit","title":"(S)GD over Diagonal Linear Networks: Implicit Regularisation, Large Stepsizes and Edge of Stability","date":"2023-02-17","arxiv_id":"2302.08982","n_code_links":0,"syntology":null},{"paper":"/paper/fosi-hybrid-first-and-second-order","slug":"fosi-hybrid-first-and-second-order","title":"FOSI: Hybrid First and Second Order Optimization","date":"2023-02-16","arxiv_id":"2302.08484","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["hsivan/fosi"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"almost-sure-saddle-avoidance-of-stochastic","title":"Almost Sure Saddle Avoidance of Stochastic Gradient Methods without the Bounded Gradient Assumption","date":"2023-02-15","arxiv_id":"2302.07862","n_code_links":0,"syntology":null},{"paper":"/paper/from-high-dimensional-mean-field-dynamics-to","slug":"from-high-dimensional-mean-field-dynamics-to","title":"From high-dimensional & mean-field dynamics to dimensionless ODEs: A unifying approach to SGD in two-layers networks","date":"2023-02-12","arxiv_id":"2302.05882","n_code_links":1,"syntology":null},{"paper":null,"slug":"cyclic-and-randomized-stepsizes-invoke","title":"Cyclic and Randomized Stepsizes Invoke Heavier Tails in SGD than Constant Stepsize","date":"2023-02-10","arxiv_id":"2302.05516","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-stochastic-gradient-4","title":"On the Convergence of Stochastic Gradient Descent for Linear Inverse Problems in Banach Spaces","date":"2023-02-10","arxiv_id":"2302.05197","n_code_links":0,"syntology":null},{"paper":null,"slug":"information-theoretic-lower-bounds-for-4","title":"Information Theoretic Lower Bounds for Information Theoretic Upper Bounds","date":"2023-02-09","arxiv_id":"2302.04925","n_code_links":0,"syntology":null}],"record_sha256":"8588b4a857765aadff27eb7029a0a9b3a7ab58aea11cfd3d2f809e529adbf849","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}