{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/2","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":21,"rows_per_page":100,"rows":[101,200],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd","next":"/method/sgd/papers/3","papers":[{"paper":null,"slug":"algorithmic-stability-of-stochastic-gradient","title":"Algorithmic Stability of Stochastic Gradient Descent with Momentum under Heavy-Tailed Noise","date":"2025-02-02","arxiv_id":"2502.00885","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-why-adam-outperforms-sgd","slug":"understanding-why-adam-outperforms-sgd","title":"Understanding Why Adam Outperforms SGD: Gradient Heterogeneity in Transformers","date":"2025-01-31","arxiv_id":"2502.00213","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-unified-analysis-of-stochastic-gradient-1","title":"A Unified Analysis of Stochastic Gradient Descent with Arbitrary Data Permutations and Beyond","date":"2025-01-27","arxiv_id":"2501.16117","n_code_links":0,"syntology":null},{"paper":null,"slug":"ringmaster-asgd-the-first-asynchronous-sgd","title":"Ringmaster ASGD: The First Asynchronous SGD with Optimal Time Complexity","date":"2025-01-27","arxiv_id":"2501.16168","n_code_links":0,"syntology":null},{"paper":null,"slug":"mathematical-analysis-of-the-gradients-in","title":"Mathematical analysis of the gradients in deep learning","date":"2025-01-26","arxiv_id":"2501.15646","n_code_links":0,"syntology":null},{"paper":null,"slug":"scalable-decentralized-learning-with","title":"Scalable Decentralized Learning with Teleportation","date":"2025-01-25","arxiv_id":"2501.15259","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-near-optimal-algorithm-for-learning-margin","title":"A Near-optimal Algorithm for Learning Margin Halfspaces with Massart Noise","date":"2025-01-16","arxiv_id":"2501.09691","n_code_links":0,"syntology":null},{"paper":null,"slug":"identification-of-traditional-medicinal-plant","title":"Identification of Traditional Medicinal Plant Leaves Using an effective Deep Learning model and Self-Curated Dataset","date":"2025-01-16","arxiv_id":"2501.09363","n_code_links":0,"syntology":null},{"paper":null,"slug":"increasing-batch-size-improves-convergence-of","title":"Increasing Batch Size Improves Convergence of Stochastic Gradient Descent with Momentum","date":"2025-01-15","arxiv_id":"2501.08883","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-stochastic-gradient-descent-effective-a","title":"Is Stochastic Gradient Descent Effective? A PDE Perspective on Machine Learning processes","date":"2025-01-14","arxiv_id":"2501.08425","n_code_links":0,"syntology":null},{"paper":null,"slug":"communication-efficient-2d-parallel","title":"Communication-Efficient, 2D Parallel Stochastic Gradient Descent for Distributed-Memory Optimization","date":"2025-01-13","arxiv_id":"2501.07526","n_code_links":0,"syntology":null},{"paper":"/paper/averaged-adam-accelerates-stochastic","slug":"averaged-adam-accelerates-stochastic","title":"Averaged Adam accelerates stochastic optimization in the training of deep neural network approximations for partial differential equation and optimal control problems","date":"2025-01-10","arxiv_id":"2501.06081","n_code_links":1,"syntology":null},{"paper":null,"slug":"revisiting-localsgd-and-scaffold-improved","title":"Revisiting LocalSGD and SCAFFOLD: Improved Rates and Missing Analysis","date":"2025-01-08","arxiv_id":"2501.04443","n_code_links":0,"syntology":null},{"paper":null,"slug":"mixing-times-and-privacy-analysis-for-the","title":"Mixing Times and Privacy Analysis for the Projected Langevin Algorithm under a Modulus of Continuity","date":"2025-01-07","arxiv_id":"2501.04134","n_code_links":0,"syntology":null},{"paper":null,"slug":"zeroflow-overcoming-catastrophic-forgetting","title":"ZeroFlow: Overcoming Catastrophic Forgetting is Easier than You Think","date":"2025-01-02","arxiv_id":"2501.01045","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-the-role-of-weight-decay-in","title":"Investigating the Role of Weight Decay in Enhancing Nonconvex SGD","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-energy-efficient-federated","title":"Accelerating Energy-Efficient Federated Learning in Cell-Free Networks with Adaptive Quantization","date":"2024-12-30","arxiv_id":"2412.20785","n_code_links":0,"syntology":null},{"paper":null,"slug":"protscan-modeling-and-prediction-of-rna","title":"ProtScan: Modeling and Prediction of RNA-Protein Interactions","date":"2024-12-30","arxiv_id":"2412.20933","n_code_links":0,"syntology":null},{"paper":null,"slug":"edge-of-stochastic-stability-revisiting-the","title":"Edge of Stochastic Stability: Revisiting the Edge of Stability for SGD","date":"2024-12-29","arxiv_id":"2412.20553","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-assembly-of-a-biologically-plausible","title":"Self-Assembly of a Biologically Plausible Learning Circuit","date":"2024-12-28","arxiv_id":"2412.20018","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-dp-sgd-with-adaptive","title":"On the Convergence of DP-SGD with Adaptive Clipping","date":"2024-12-27","arxiv_id":"2412.19916","n_code_links":0,"syntology":null},{"paper":null,"slug":"torque-aware-momentum","title":"Torque-Aware Momentum","date":"2024-12-25","arxiv_id":"2412.18790","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-generate-gradients-for-test-time","slug":"learning-to-generate-gradients-for-test-time","title":"Learning to Generate Gradients for Test-Time Adaptation via Test-Time Training Layers","date":"2024-12-22","arxiv_id":"2412.16901","n_code_links":1,"syntology":null},{"paper":null,"slug":"gradient-based-non-linear-inverse-learning","title":"Gradient-Based Non-Linear Inverse Learning","date":"2024-12-21","arxiv_id":"2412.16794","n_code_links":0,"syntology":null},{"paper":null,"slug":"foxtsage-vs-adam-revolution-or-evolution-in","title":"Foxtsage vs. Adam: Revolution or Evolution in Optimization?","date":"2024-12-20","arxiv_id":"2412.17855","n_code_links":0,"syntology":null},{"paper":null,"slug":"swan-preprocessing-sgd-enables-adam-level","title":"SWAN: SGD with Normalization and Whitening Enables Stateless LLM Training","date":"2024-12-17","arxiv_id":"2412.13148","n_code_links":0,"syntology":null},{"paper":"/paper/explicit-and-implicit-graduated-optimization","slug":"explicit-and-implicit-graduated-optimization","title":"Explicit and Implicit Graduated Optimization in Deep Neural Networks","date":"2024-12-16","arxiv_id":"2412.11501","n_code_links":1,"syntology":null},{"paper":"/paper/no-more-adam-learning-rate-scaling-at","slug":"no-more-adam-learning-rate-scaling-at","title":"No More Adam: Learning Rate Scaling at Initialization is All You Need","date":"2024-12-16","arxiv_id":"2412.11768","n_code_links":1,"syntology":null},{"paper":"/paper/coupling-based-convergence-diagnostic-and","slug":"coupling-based-convergence-diagnostic-and","title":"Coupling-based Convergence Diagnostic and Stepsize Scheme for Stochastic Gradient Descent","date":"2024-12-15","arxiv_id":"2412.11341","n_code_links":1,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-in-the-optimal","title":"Stochastic Gradient Descent in the Optimal Control of Execution Costs","date":"2024-12-14","arxiv_id":"2412.12199","n_code_links":0,"syntology":null},{"paper":null,"slug":"understand-the-effectiveness-of-shortcuts","title":"Understand the Effectiveness of Shortcuts through the Lens of DCA","date":"2024-12-13","arxiv_id":"2412.09853","n_code_links":0,"syntology":null},{"paper":"/paper/edit-a-local-sgd-based-efficient-distributed","slug":"edit-a-local-sgd-based-efficient-distributed","title":"EDiT: A Local-SGD-Based Efficient Distributed Training Method for Large Language Models","date":"2024-12-10","arxiv_id":"2412.07210","n_code_links":2,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-revisited","title":"Stochastic Gradient Descent Revisited","date":"2024-12-08","arxiv_id":"2412.06070","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-optimization-for-enhanced-efficiency","title":"Adaptive Optimization for Enhanced Efficiency in Large-Scale Language Model Training","date":"2024-12-06","arxiv_id":"2412.04718","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-granger-causal-perspective-on-gradient","title":"A Granger-Causal Perspective on Gradient Descent with Application to Pruning","date":"2024-12-04","arxiv_id":"2412.03035","n_code_links":0,"syntology":null},{"paper":null,"slug":"machine-learning-methods-for-automated","title":"Machine Learning Methods for Automated Interstellar Object Classification with LSST","date":"2024-12-03","arxiv_id":"2412.02112","n_code_links":0,"syntology":null},{"paper":null,"slug":"memory-efficient-training-for-deep-speaker","title":"Memory-Efficient Training for Deep Speaker Embedding Learning in Speaker Verification","date":"2024-12-02","arxiv_id":"2412.01195","n_code_links":0,"syntology":null},{"paper":null,"slug":"profit-a-proximal-fine-tuning-optimizer-for","title":"PROFIT: A Specialized Optimizer for Deep Fine Tuning","date":"2024-12-02","arxiv_id":"2412.01930","n_code_links":0,"syntology":null},{"paper":null,"slug":"training-multi-layer-binary-neural-networks","title":"Training Multi-Layer Binary Neural Networks With Local Binary Error Signals","date":"2024-11-28","arxiv_id":"2412.00119","n_code_links":0,"syntology":null},{"paper":null,"slug":"exponential-moving-average-of-weights-in-deep","title":"Exponential Moving Average of Weights in Deep Learning: Dynamics and Benefits","date":"2024-11-27","arxiv_id":"2411.18704","n_code_links":0,"syntology":null},{"paper":null,"slug":"lion-cub-minimizing-communication-overhead-in","title":"Lion Cub: Minimizing Communication Overhead in Distributed Lion","date":"2024-11-25","arxiv_id":"2411.16462","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-methods-through-the-lens-of-sdes","title":"Adaptive Methods through the Lens of SDEs: Theoretical Insights on the Role of Noise","date":"2024-11-24","arxiv_id":"2411.15958","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-unified-analysis-for-finite-weight","title":"A Unified Analysis for Finite Weight Averaging","date":"2024-11-20","arxiv_id":"2411.13169","n_code_links":0,"syntology":null},{"paper":null,"slug":"classification-of-geographical-land-structure","title":"Classification of Geographical Land Structure Using Convolution Neural Network and Transfer Learning","date":"2024-11-19","arxiv_id":"2411.12415","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-rate-analysis-of-lion","title":"Convergence Rate Analysis of LION","date":"2024-11-12","arxiv_id":"2411.07724","n_code_links":0,"syntology":null},{"paper":"/paper/frugal-memory-efficient-optimization-by","slug":"frugal-memory-efficient-optimization-by","title":"FRUGAL: Memory-Efficient Optimization by Reducing State Overhead for Scalable Training","date":"2024-11-12","arxiv_id":"2411.07837","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":4,"n_instrument":4,"unverified":0,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fzmushko/frugal"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"efficient-adaptive-optimization-via-subset","title":"Lean and Mean Adaptive Optimization via Subset-Norm and Subspace-Momentum with Convergence Guarantees","date":"2024-11-11","arxiv_id":"2411.07120","n_code_links":0,"syntology":null},{"paper":null,"slug":"general-framework-for-online-to-nonconvex","title":"General framework for online-to-nonconvex conversion: Schedule-free SGD is also effective for nonconvex optimization","date":"2024-11-11","arxiv_id":"2411.07061","n_code_links":0,"syntology":null},{"paper":null,"slug":"impact-of-label-noise-on-learning-complex","title":"Impact of Label Noise on Learning Complex Features","date":"2024-11-07","arxiv_id":"2411.04569","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-bayesian-approach-to-data-point-selection","title":"A Bayesian Approach to Data Point Selection","date":"2024-11-06","arxiv_id":"2411.03768","n_code_links":0,"syntology":null},{"paper":null,"slug":"point-processes-with-event-time-uncertainty","title":"Point processes with event time uncertainty","date":"2024-11-05","arxiv_id":"2411.02694","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-and-transferring-sparse-contextual","title":"Learning and Transferring Sparse Contextual Bigrams with Linear Transformers","date":"2024-10-30","arxiv_id":"2410.23438","n_code_links":0,"syntology":null},{"paper":null,"slug":"developing-convolutional-neural-networks","title":"Developing Convolutional Neural Networks using a Novel Lamarckian Co-Evolutionary Algorithm","date":"2024-10-29","arxiv_id":"2410.22487","n_code_links":0,"syntology":null},{"paper":null,"slug":"shuffling-gradient-based-methods-for","title":"Shuffling Gradient-Based Methods for Nonconvex-Concave Minimax Optimization","date":"2024-10-29","arxiv_id":"2410.22297","n_code_links":0,"syntology":null},{"paper":null,"slug":"trustworthiness-of-stochastic-gradient","title":"Trustworthiness of Stochastic Gradient Descent in Distributed Learning","date":"2024-10-28","arxiv_id":"2410.21491","n_code_links":0,"syntology":null},{"paper":"/paper/differentially-private-learning-needs-better","slug":"differentially-private-learning-needs-better","title":"Differentially Private Learning Needs Better Model Initialization and Self-Distillation","date":"2024-10-23","arxiv_id":"2410.17566","n_code_links":1,"syntology":null},{"paper":null,"slug":"gradient-normalization-with-out-clipping","title":"Gradient Normalization Provably Benefits Nonconvex SGD under Heavy-Tailed Noise","date":"2024-10-21","arxiv_id":"2410.16561","n_code_links":0,"syntology":null},{"paper":null,"slug":"simplicity-bias-via-global-convergence-of","title":"Simplicity Bias via Global Convergence of Sharpness Minimization","date":"2024-10-21","arxiv_id":"2410.16401","n_code_links":0,"syntology":null},{"paper":null,"slug":"unleashing-the-potential-of-multi-channel","title":"Unleashing the Potential of Multi-Channel Fusion in Retrieval for Personalized Recommendations","date":"2024-10-21","arxiv_id":"2410.16080","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-discretization-learning-the-optimal","slug":"beyond-discretization-learning-the-optimal","title":"Beyond Discretization: Learning the Optimal Solution Path","date":"2024-10-18","arxiv_id":"2410.14885","n_code_links":1,"syntology":null},{"paper":null,"slug":"implicit-regularization-of-sharpness-aware","title":"Implicit Regularization of Sharpness-Aware Minimization for Scale-Invariant Problems","date":"2024-10-18","arxiv_id":"2410.14802","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-jittering-for","title":"SGD Jittering: A Training Strategy for Robust and Accurate Model-Based Architectures","date":"2024-10-18","arxiv_id":"2410.14667","n_code_links":0,"syntology":null},{"paper":"/paper/active-dormant-attention-heads","slug":"active-dormant-attention-heads","title":"Active-Dormant Attention Heads: Mechanistically Demystifying Extreme-Token Phenomena in LLMs","date":"2024-10-17","arxiv_id":"2410.13835","n_code_links":1,"syntology":{"ran":1,"of":6,"n_ran_checked":1,"n_instrument":0,"unverified":5,"pointer_only":6,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["guotianyu2000/active-dormant-attention"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-self-generated-documents-for","title":"Evaluating Self-Generated Documents for Enhancing Retrieval-Augmented Generation with Large Language Models","date":"2024-10-17","arxiv_id":"2410.13192","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-gradient-clipping-to-normalization-for","title":"From Gradient Clipping to Normalization for Heavy Tailed SGD","date":"2024-10-17","arxiv_id":"2410.13849","n_code_links":0,"syntology":null},{"paper":"/paper/mitigating-hallucinations-in-large-vision-2","slug":"mitigating-hallucinations-in-large-vision-2","title":"Mitigating Hallucinations in Large Vision-Language Models via Summary-Guided Decoding","date":"2024-10-17","arxiv_id":"2410.13321","n_code_links":1,"syntology":null},{"paper":null,"slug":"nonlinear-stochastic-gradient-descent-and","title":"Nonlinear Stochastic Gradient Descent and Heavy-tailed Noise: A Unified Framework and High-probability Guarantees","date":"2024-10-17","arxiv_id":"2410.13954","n_code_links":0,"syntology":null},{"paper":null,"slug":"advanced-persistent-threats-apt-attribution","title":"Advanced Persistent Threats (APT) Attribution Using Deep Reinforcement Learning","date":"2024-10-15","arxiv_id":"2410.11463","n_code_links":0,"syntology":null},{"paper":null,"slug":"age-of-gradient-updates-for-federated","title":"Age-of-Gradient Updates for Federated Learning over Random Access Channels","date":"2024-10-15","arxiv_id":"2410.11986","n_code_links":0,"syntology":null},{"paper":null,"slug":"non-convergence-to-global-minimizers-in-data","title":"Non-convergence to global minimizers in data driven supervised deep learning: Adam and stochastic gradient descent optimization provably fail to converge to global minimizers in the training of deep neural networks with ReLU activation","date":"2024-10-14","arxiv_id":"2410.10533","n_code_links":0,"syntology":null},{"paper":"/paper/sampa-sharpness-aware-minimization","slug":"sampa-sharpness-aware-minimization","title":"SAMPa: Sharpness-aware Minimization Parallelized","date":"2024-10-14","arxiv_id":"2410.10683","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":13,"n_instrument":0,"unverified":2,"pointer_only":15,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["lions-epfl/sampa"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/sharpness-aware-minimization-efficiently","slug":"sharpness-aware-minimization-efficiently","title":"Sharpness-Aware Minimization Efficiently Selects Flatter Minima Late in Training","date":"2024-10-14","arxiv_id":"2410.10373","n_code_links":0,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"learning-orthogonal-multi-index-models-a-fine","title":"Learning Orthogonal Multi-Index Models: A Fine-Grained Information Exponent Analysis","date":"2024-10-13","arxiv_id":"2410.09678","n_code_links":0,"syntology":null},{"paper":null,"slug":"sharper-guarantees-for-learning-neural","title":"Sharper Guarantees for Learning Neural Network Classifiers with Gradient Methods","date":"2024-10-13","arxiv_id":"2410.10024","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-deletion-for-linear-regression-with","title":"Data Deletion for Linear Regression with Noisy SGD","date":"2024-10-12","arxiv_id":"2410.09311","n_code_links":0,"syntology":null},{"paper":"/paper/subzero-random-subspace-zeroth-order","slug":"subzero-random-subspace-zeroth-order","title":"Zeroth-Order Fine-Tuning of LLMs in Random Subspaces","date":"2024-10-11","arxiv_id":"2410.08989","n_code_links":1,"syntology":{"ran":11,"of":15,"n_ran_checked":6,"n_instrument":5,"unverified":4,"pointer_only":15,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","official":{"repos":["zimingyy/subzero"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/adam-exploits-ell-infty-geometry-of-loss","slug":"adam-exploits-ell-infty-geometry-of-loss","title":"Adam Exploits $\\ell_\\infty$-geometry of Loss Landscape via Coordinate-wise Adaptivity","date":"2024-10-10","arxiv_id":"2410.08198","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":6,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["mohamad-amin/adam-coordinate-adaptivity"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-the-convergence-of-stochastic-gradient-6","title":"On the Convergence of (Stochastic) Gradient Descent for Kolmogorov--Arnold Networks","date":"2024-10-10","arxiv_id":"2410.08041","n_code_links":0,"syntology":null},{"paper":"/paper/a-second-order-like-optimizer-with-adaptive","slug":"a-second-order-like-optimizer-with-adaptive","title":"A second-order-like optimizer with adaptive gradient scaling for deep learning","date":"2024-10-08","arxiv_id":"2410.05871","n_code_links":1,"syntology":null},{"paper":null,"slug":"asynchronous-stochastic-gradient-descent-with-2","title":"Asynchronous Stochastic Gradient Descent with Decoupled Backpropagation and Layer-Wise Updates","date":"2024-10-08","arxiv_id":"2410.05985","n_code_links":0,"syntology":null},{"paper":null,"slug":"unveiling-the-backbone-optimizer-coupling","title":"Unveiling the Backbone-Optimizer Coupling Bias in Visual Representation Learning","date":"2024-10-08","arxiv_id":"2410.06373","n_code_links":0,"syntology":null},{"paper":null,"slug":"nonasymptotic-analysis-of-stochastic-gradient-1","title":"Nonasymptotic Analysis of Stochastic Gradient Descent with the Richardson-Romberg Extrapolation","date":"2024-10-07","arxiv_id":"2410.05106","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comprehensive-framework-for-analyzing-the","title":"A Comprehensive Framework for Analyzing the Convergence of Adam: Bridging the Gap with SGD","date":"2024-10-06","arxiv_id":"2410.04458","n_code_links":0,"syntology":null},{"paper":null,"slug":"mindflayer-efficient-asynchronous-parallel","title":"MindFlayer SGD: Efficient Parallel SGD in the Presence of Heterogeneous and Random Worker Compute Times","date":"2024-10-05","arxiv_id":"2410.04285","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-with-memory-fundamental-properties-and","title":"SGD with memory: fundamental properties and stochastic acceleration","date":"2024-10-05","arxiv_id":"2410.04228","n_code_links":0,"syntology":null},{"paper":null,"slug":"collaborative-and-efficient-personalization","title":"Collaborative and Efficient Personalization with Mixtures of Adaptors","date":"2024-10-04","arxiv_id":"2410.03497","n_code_links":0,"syntology":null},{"paper":"/paper/estimating-generalization-performance-along","slug":"estimating-generalization-performance-along","title":"Estimating Generalization Performance Along the Trajectory of Proximal SGD in Robust Regression","date":"2024-10-03","arxiv_id":"2410.02629","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-better-generalization-weight-decay","title":"Towards Better Generalization: Weight Decay Induces Low-rank Bias for Neural Networks","date":"2024-10-03","arxiv_id":"2410.02176","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-with-adaptive","title":"Stochastic Gradient Descent with Adaptive Data","date":"2024-10-02","arxiv_id":"2410.01195","n_code_links":0,"syntology":null},{"paper":null,"slug":"truncated-kernel-stochastic-gradient-descent","title":"Truncated Kernel Stochastic Gradient Descent on Spheres","date":"2024-10-02","arxiv_id":"2410.01570","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-unidirectional-bidirectional-and","title":"Comparing Unidirectional, Bidirectional, and Word2vec Models for Discovering Vulnerabilities in Compiled Lifted Code","date":"2024-09-26","arxiv_id":"2409.17513","n_code_links":0,"syntology":null},{"paper":null,"slug":"does-worst-performing-agent-lead-the-pack","title":"Does Worst-Performing Agent Lead the Pack? Analyzing Agent Dynamics in Unified Distributed SGD","date":"2024-09-26","arxiv_id":"2409.17499","n_code_links":0,"syntology":null},{"paper":null,"slug":"differential-privacy-regularization","title":"Differential Privacy Regularization: Protecting Training Data Through Loss Function Regularization","date":"2024-09-25","arxiv_id":"2409.17144","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-of-distributed-adaptive","title":"Convergence of Distributed Adaptive Optimization with Local Updates","date":"2024-09-20","arxiv_id":"2409.13155","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-scaling-laws-for-local-sgd-in-large","title":"Exploring Scaling Laws for Local SGD in Large Language Model Training","date":"2024-09-20","arxiv_id":"2409.13198","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-optimality-of-accelerated-sgd-for-high","title":"The Optimality of (Accelerated) SGD for High-Dimensional Quadratic Optimization","date":"2024-09-15","arxiv_id":"2409.09745","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-dynamic-weighting-strategy-to-mitigate","title":"A Dynamic Weighting Strategy to Mitigate Worker Node Failure in Distributed Deep Learning","date":"2024-09-14","arxiv_id":"2409.09242","n_code_links":0,"syntology":null},{"paper":"/paper/increasing-both-batch-size-and-learning-rate","slug":"increasing-both-batch-size-and-learning-rate","title":"Increasing Both Batch Size and Learning Rate Accelerates Stochastic Gradient Descent","date":"2024-09-13","arxiv_id":"2409.08770","n_code_links":1,"syntology":null},{"paper":"/paper/asymptotics-of-stochastic-gradient-descent","slug":"asymptotics-of-stochastic-gradient-descent","title":"Asymptotics of Stochastic Gradient Descent with Dropout Regularization in Linear Models","date":"2024-09-11","arxiv_id":"2409.07434","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-deep-learning-with-optimized","title":"Enhancing Deep Learning with Optimized Gradient Descent: Bridging Numerical Methods and Neural Network Training","date":"2024-09-07","arxiv_id":"2409.04707","n_code_links":0,"syntology":null}],"record_sha256":"9a44706ee7d6b4309b791459c6c4893409ba681191b8458bcce408be38632c86","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}