{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/3","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":3,"pages_in_order":21,"rows_per_page":100,"rows":[201,300],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/2","next":"/method/sgd/papers/4","papers":[{"paper":null,"slug":"fast-forwarding-low-rank-training","title":"Fast Forwarding Low-Rank Training","date":"2024-09-06","arxiv_id":"2409.04206","n_code_links":0,"syntology":null},{"paper":null,"slug":"bootstrap-sgd-algorithmic-stability-and","title":"Bootstrap SGD: Algorithmic Stability and Robustness","date":"2024-09-02","arxiv_id":"2409.01074","n_code_links":0,"syntology":null},{"paper":"/paper/1-bit-fqt-pushing-the-limit-of-fully","slug":"1-bit-fqt-pushing-the-limit-of-fully","title":"1-Bit FQT: Pushing the Limit of Fully Quantized Training to 1-bit","date":"2024-08-26","arxiv_id":"2408.14267","n_code_links":1,"syntology":null},{"paper":"/paper/pfdiff-training-free-acceleration-of","slug":"pfdiff-training-free-acceleration-of","title":"PFDiff: Training-free Acceleration of Diffusion Models through the Gradient Guidance of Past and Future","date":"2024-08-16","arxiv_id":"2408.08822","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":2,"n_instrument":6,"unverified":1,"pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","official":{"repos":["onefly123/PFDiff"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"high-dimensional-optimization-for-multi","title":"Langevin dynamics for high-dimensional optimization: the case of multi-spiked tensor PCA","date":"2024-08-12","arxiv_id":"2408.06401","n_code_links":0,"syntology":null},{"paper":"/paper/incremental-gauss-newton-descent-for-machine","slug":"incremental-gauss-newton-descent-for-machine","title":"Incremental Gauss-Newton Descent for Machine Learning","date":"2024-08-10","arxiv_id":"2408.05560","n_code_links":1,"syntology":null},{"paper":null,"slug":"mathematical-programming-for-adaptive","title":"Optimization-Driven Adaptive Experimentation","date":"2024-08-08","arxiv_id":"2408.04570","n_code_links":0,"syntology":null},{"paper":"/paper/2408-02654","slug":"2408-02654","title":"On Using Quasirandom Sequences in Machine Learning for Model Weight Initialization","date":"2024-08-05","arxiv_id":"2408.02654","n_code_links":1,"syntology":null},{"paper":null,"slug":"2408-02839","title":"Optimizing Cox Models with Stochastic Gradient Descent: Theoretical Foundations and Practical Guidances","date":"2024-08-05","arxiv_id":"2408.02839","n_code_links":0,"syntology":null},{"paper":"/paper/2408-01638","slug":"2408-01638","title":"Transforming Slot Schema Induction with Generative Dialogue State Inference","date":"2024-08-03","arxiv_id":"2408.01638","n_code_links":1,"syntology":null},{"paper":null,"slug":"2408-01736","title":"Can LLMs predict the convergence of Stochastic Gradient Descent?","date":"2024-08-03","arxiv_id":"2408.01736","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-01365","title":"Data Debugging is NP-hard for Classifiers Trained with SGD","date":"2024-08-02","arxiv_id":"2408.01365","n_code_links":0,"syntology":null},{"paper":"/paper/2407-21633","slug":"2407-21633","title":"Zero-Shot Cross-Domain Dialogue State Tracking via Dual Low-Rank Adaptation","date":"2024-07-31","arxiv_id":"2407.21633","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":6,"n_instrument":1,"unverified":3,"pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["suntea233/duallora"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/no-learning-rates-needed-introducing-salsa","slug":"no-learning-rates-needed-introducing-salsa","title":"No learning rates needed: Introducing SALSA -- Stable Armijo Line Search Adaptation","date":"2024-07-30","arxiv_id":"2407.20650","n_code_links":1,"syntology":null},{"paper":"/paper/the-entrapment-problem-in-random-walk","slug":"the-entrapment-problem-in-random-walk","title":"The Entrapment Problem in Random Walk Decentralized Learning","date":"2024-07-30","arxiv_id":"2407.20611","n_code_links":1,"syntology":null},{"paper":null,"slug":"2407-21078","title":"Convergence rates for the Adam optimizer","date":"2024-07-29","arxiv_id":"2407.21078","n_code_links":0,"syntology":null},{"paper":null,"slug":"characterizing-dynamical-stability-of","title":"Characterizing Dynamical Stability of Stochastic Gradient Descent in Overparameterized Learning","date":"2024-07-29","arxiv_id":"2407.20209","n_code_links":0,"syntology":null},{"paper":null,"slug":"ordered-momentum-for-asynchronous-sgd","title":"Ordered Momentum for Asynchronous SGD","date":"2024-07-27","arxiv_id":"2407.19234","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-theoretical-perspective-on-data","title":"A New Theoretical Perspective on Data Heterogeneity in Federated Optimization","date":"2024-07-22","arxiv_id":"2407.15567","n_code_links":0,"syntology":null},{"paper":null,"slug":"natural-language-task-oriented-dialog-system","title":"Training Zero-Shot Generalizable End-to-End Task-Oriented Dialog System Without Turn-level Dialog Annotations","date":"2024-07-21","arxiv_id":"2407.15055","n_code_links":0,"syntology":null},{"paper":null,"slug":"non-convergence-of-adam-and-other-adaptive","title":"Non-convergence of Adam and other adaptive stochastic gradient descent optimization methods for non-vanishing learning rates","date":"2024-07-11","arxiv_id":"2407.08100","n_code_links":0,"syntology":null},{"paper":null,"slug":"predicting-heart-failure-with-attention","title":"Predicting Heart Failure with Attention Learning Techniques Utilizing Cardiovascular Data","date":"2024-07-11","arxiv_id":"2407.08289","n_code_links":0,"syntology":null},{"paper":null,"slug":"deconstructing-what-makes-a-good-optimizer","title":"Deconstructing What Makes a Good Optimizer for Language Models","date":"2024-07-10","arxiv_id":"2407.07972","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-for-two-layer","title":"Stochastic Gradient Descent for Two-layer Neural Networks","date":"2024-07-10","arxiv_id":"2407.07670","n_code_links":0,"syntology":null},{"paper":"/paper/loco-low-bit-communication-adaptor-for-large","slug":"loco-low-bit-communication-adaptor-for-large","title":"LoCo: Low-Bit Communication Adaptor for Large-scale Model Training","date":"2024-07-05","arxiv_id":"2407.04480","n_code_links":1,"syntology":null},{"paper":null,"slug":"bias-of-stochastic-gradient-descent-or-the","title":"Bias of Stochastic Gradient Descent or the Architecture: Disentangling the Effects of Overparameterization of Neural Networks","date":"2024-07-04","arxiv_id":"2407.03848","n_code_links":0,"syntology":null},{"paper":"/paper/automatic-gradient-descent-with-generalized","slug":"automatic-gradient-descent-with-generalized","title":"Gradient descent with generalized Newton's method","date":"2024-07-03","arxiv_id":"2407.02772","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shiyunxu/autogen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"universal-length-generalization-with-turing","title":"Universal Length Generalization with Turing Programs","date":"2024-07-03","arxiv_id":"2407.03310","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-differential-equations-models-for","title":"Stochastic Differential Equations models for Least-Squares Stochastic Gradient Descent","date":"2024-07-02","arxiv_id":"2407.02322","n_code_links":0,"syntology":null},{"paper":null,"slug":"infinite-width-models-that-work-why-feature","title":"Infinite Width Models That Work: Why Feature Learning Doesn't Matter as Much as You Think","date":"2024-06-27","arxiv_id":"2406.18800","n_code_links":0,"syntology":null},{"paper":"/paper/towards-efficient-and-scalable-training-of","slug":"towards-efficient-and-scalable-training-of","title":"Towards Efficient and Scalable Training of Differentially Private Deep Learning","date":"2024-06-25","arxiv_id":"2406.17298","n_code_links":1,"syntology":null},{"paper":null,"slug":"effect-of-random-learning-rate-theoretical","title":"Effect of Random Learning Rate: Theoretical Analysis of SGD Dynamics in Non-Convex Optimization via Stationary Distribution","date":"2024-06-23","arxiv_id":"2406.16032","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-batch-analysis-for-adagrad-under","title":"AdaGrad under Anisotropic Smoothness","date":"2024-06-21","arxiv_id":"2406.15244","n_code_links":0,"syntology":null},{"paper":null,"slug":"communication-efficient-adaptive-batch-size","title":"Communication-Efficient Adaptive Batch Size Strategies for Distributed Local Gradient Methods","date":"2024-06-20","arxiv_id":"2406.13936","n_code_links":0,"syntology":null},{"paper":"/paper/learning-rate-adaptive-stochastic-gradient","slug":"learning-rate-adaptive-stochastic-gradient","title":"Learning rate adaptive stochastic gradient descent optimization methods: numerical simulations for deep learning methods for partial differential equations and convergence analyses","date":"2024-06-20","arxiv_id":"2406.14340","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-exact-gradient-based-training-on","title":"Towards Exact Gradient-based Training on Analog In-memory Computing","date":"2024-06-18","arxiv_id":"2406.12774","n_code_links":0,"syntology":null},{"paper":"/paper/a-clipped-trip-the-dynamics-of-sgd-with","slug":"a-clipped-trip-the-dynamics-of-sgd-with","title":"To Clip or not to Clip: the Dynamics of SGD with Gradient Clipping in High-Dimensions","date":"2024-06-17","arxiv_id":"2406.11733","n_code_links":1,"syntology":null},{"paper":null,"slug":"distributed-stochastic-gradient-descent-with-1","title":"Distributed Stochastic Gradient Descent with Staleness: A Stochastic Delay Differential Equation Based Framework","date":"2024-06-17","arxiv_id":"2406.11159","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-neural-networks-learn-the-support-is-an","title":"How Neural Networks Learn the Support is an Implicit Regularization Effect of SGD","date":"2024-06-17","arxiv_id":"2406.11110","n_code_links":0,"syntology":null},{"paper":null,"slug":"just-how-flexible-are-neural-networks-in","title":"Just How Flexible are Neural Networks in Practice?","date":"2024-06-17","arxiv_id":"2406.11463","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-is-the-long-run-distribution-of","title":"What is the long-run distribution of stochastic gradient descent? A large deviations analysis","date":"2024-06-13","arxiv_id":"2406.09241","n_code_links":0,"syntology":null},{"paper":"/paper/why-warmup-the-learning-rate-underlying","slug":"why-warmup-the-learning-rate-underlying","title":"Why Warmup the Learning Rate? Underlying Mechanisms and Improvements","date":"2024-06-13","arxiv_id":"2406.09405","n_code_links":0,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"scaling-laws-in-linear-regression-compute","title":"Scaling Laws in Linear Regression: Compute, Parameters, and Data","date":"2024-06-12","arxiv_id":"2406.08466","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-analysis-of-adaptive-gradient","title":"Provable Complexity Improvement of AdaGrad over SGD: Upper and Lower Bounds in Stochastic Non-Convex Optimization","date":"2024-06-07","arxiv_id":"2406.04592","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-expanding-scope-of-the-stability-gap","title":"The Expanding Scope of the Stability Gap: Unveiling its Presence in Joint Incremental Learning of Homogeneous Tasks","date":"2024-06-07","arxiv_id":"2406.05114","n_code_links":0,"syntology":null},{"paper":null,"slug":"concurrent-training-and-layer-pruning-of-deep","title":"Concurrent Training and Layer Pruning of Deep Neural Networks","date":"2024-06-06","arxiv_id":"2406.04549","n_code_links":0,"syntology":null},{"paper":"/paper/stochastic-polyak-step-sizes-and-momentum","slug":"stochastic-polyak-step-sizes-and-momentum","title":"Stochastic Polyak Step-sizes and Momentum: Convergence Guarantees and Practical Performance","date":"2024-06-06","arxiv_id":"2406.04142","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["dimitris-oik/momsps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/online-learning-and-information-exponents-on","slug":"online-learning-and-information-exponents-on","title":"Online Learning and Information Exponents: On The Importance of Batch size, and Time/Complexity Tradeoffs","date":"2024-06-04","arxiv_id":"2406.02157","n_code_links":1,"syntology":null},{"paper":null,"slug":"cohort-squeeze-beyond-a-single-communication","title":"Cohort Squeeze: Beyond a Single Communication Round per Cohort in Cross-Device Federated Learning","date":"2024-06-03","arxiv_id":"2406.01115","n_code_links":0,"syntology":null},{"paper":null,"slug":"demystifying-sgd-with-doubly-stochastic","title":"Demystifying SGD with Doubly Stochastic Gradients","date":"2024-06-03","arxiv_id":"2406.00920","n_code_links":0,"syntology":null},{"paper":null,"slug":"local-methods-with-adaptivity-via-scaling","title":"Local Methods with Adaptivity via Scaling","date":"2024-06-02","arxiv_id":"2406.00846","n_code_links":0,"syntology":null},{"paper":null,"slug":"machine-learning-methods-for-pricing","title":"Machine Learning Methods for Pricing Financial Derivatives","date":"2024-06-01","arxiv_id":"2406.00459","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-restarting-to-overcome-overfitting","title":"Stochastic Resetting Mitigates Latent Gradient Bias of SGD from Label Noise","date":"2024-06-01","arxiv_id":"2406.00396","n_code_links":0,"syntology":null},{"paper":"/paper/pixood-pixel-level-out-of-distribution","slug":"pixood-pixel-level-out-of-distribution","title":"PixOOD: Pixel-Level Out-of-Distribution Detection","date":"2024-05-30","arxiv_id":"2405.19882","n_code_links":1,"syntology":null},{"paper":null,"slug":"sharpness-aware-minimization-enhances-feature","title":"Sharpness-Aware Minimization Enhances Feature Quality via Balanced Learning","date":"2024-05-30","arxiv_id":"2405.20439","n_code_links":0,"syntology":null},{"paper":null,"slug":"symmetries-in-overparametrized-neural","title":"Symmetries in Overparametrized Neural Networks: A Mean-Field View","date":"2024-05-30","arxiv_id":"2405.19995","n_code_links":0,"syntology":null},{"paper":"/paper/the-high-line-exact-risk-and-learning-rate","slug":"the-high-line-exact-risk-and-learning-rate","title":"The High Line: Exact Risk and Learning Rate Curves of Stochastic Adaptive Learning Rate Algorithms","date":"2024-05-30","arxiv_id":"2405.19585","n_code_links":1,"syntology":null},{"paper":null,"slug":"spaba-a-single-loop-and-probabilistic","title":"SPABA: A Single-Loop and Probabilistic Stochastic Bilevel Algorithm Achieving Optimal Sample Complexity","date":"2024-05-29","arxiv_id":"2405.18777","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-hessian-aware-stochastic-differential","title":"A Hessian-Aware Stochastic Differential Equation for Modelling SGD","date":"2024-05-28","arxiv_id":"2405.18373","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-margin-based-multiclass-generalization","title":"A Margin-based Multiclass Generalization Bound via Geometric Complexity","date":"2024-05-28","arxiv_id":"2405.18590","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-unified-balance-theory-of-second-moment","title":"The Unified Balance Theory of Second-Moment Exponential Scaling Optimizers in Visual Tasks","date":"2024-05-28","arxiv_id":"2405.18498","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-communication-efficient-federated","title":"Towards Communication-efficient Federated Learning via Sparse and Aligned Adaptive Optimization","date":"2024-05-28","arxiv_id":"2405.17932","n_code_links":0,"syntology":null},{"paper":null,"slug":"dual-delayed-asynchronous-sgd-for-arbitrarily","title":"Dual-Delayed Asynchronous SGD for Arbitrarily Heterogeneous Data","date":"2024-05-27","arxiv_id":"2405.16966","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-does-perfect-fitting-affect","title":"How Do the Architecture and Optimizer Affect Representation Learning? On the Training Dynamics of Representations in Deep Neural Networks","date":"2024-05-27","arxiv_id":"2405.17377","n_code_links":0,"syntology":null},{"paper":"/paper/adafisher-adaptive-second-order-optimization","slug":"adafisher-adaptive-second-order-optimization","title":"AdaFisher: Adaptive Second Order Optimization via Fisher Information","date":"2024-05-26","arxiv_id":"2405.16397","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":2,"n_instrument":5,"unverified":3,"pointer_only":10,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","official":{"repos":["AtlasAnalyticsLab/AdaFisher"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/does-sgd-really-happen-in-tiny-subspaces","slug":"does-sgd-really-happen-in-tiny-subspaces","title":"Does SGD really happen in tiny subspaces?","date":"2024-05-25","arxiv_id":"2405.16002","n_code_links":0,"syntology":{"ran":14,"of":15,"n_ran_checked":13,"n_instrument":1,"unverified":1,"pointer_only":15,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"derivatives-of-stochastic-gradient-descent","title":"Derivatives of Stochastic Gradient Descent in parametric optimization","date":"2024-05-24","arxiv_id":"2405.15894","n_code_links":0,"syntology":null},{"paper":null,"slug":"freya-page-first-optimal-time-complexity-for","title":"Freya PAGE: First Optimal Time Complexity for Large-Scale Nonconvex Finite-Sum Optimization with Heterogeneous Asynchronous Computations","date":"2024-05-24","arxiv_id":"2405.15545","n_code_links":0,"syntology":null},{"paper":"/paper/exact-gauss-newton-optimization-for-training","slug":"exact-gauss-newton-optimization-for-training","title":"Exact Gauss-Newton Optimization for Training Deep Neural Networks","date":"2024-05-23","arxiv_id":"2405.14402","n_code_links":1,"syntology":null},{"paper":null,"slug":"semi-discrete-optimal-transport-nearly","title":"Semi-Discrete Optimal Transport: Nearly Minimax Estimation With Stochastic Gradient Descent and Adaptive Entropic Regularization","date":"2024-05-23","arxiv_id":"2405.14459","n_code_links":0,"syntology":null},{"paper":null,"slug":"surge-phenomenon-in-optimal-learning-rate-and","title":"Surge Phenomenon in Optimal Learning Rate and Batch Size Scaling","date":"2024-05-23","arxiv_id":"2405.14578","n_code_links":0,"syntology":null},{"paper":null,"slug":"score-based-generative-models-with-adaptive","title":"Score-based Generative Models with Adaptive Momentum","date":"2024-05-22","arxiv_id":"2405.13726","n_code_links":0,"syntology":null},{"paper":null,"slug":"uncertainty-quantification-by-block-bootstrap","title":"Uncertainty quantification by block bootstrap for differentially private stochastic gradient descent","date":"2024-05-21","arxiv_id":"2405.12553","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-limits-and-potentials-of-local-sgd-for","title":"The Limits and Potentials of Local SGD for Distributed Heterogeneous Learning with Intermittent Communication","date":"2024-05-19","arxiv_id":"2405.11667","n_code_links":0,"syntology":null},{"paper":"/paper/audio-visual-speech-recognition-based-on","slug":"audio-visual-speech-recognition-based-on","title":"Audio-Visual Speech Recognition based on Regulated Transformer and Spatio-Temporal Fusion Strategy for Driver Assistive Systems","date":"2024-05-09","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"custom-gradient-estimators-are-straight","title":"Custom Gradient Estimators are Straight-Through Estimators in Disguise","date":"2024-05-08","arxiv_id":"2405.05171","n_code_links":0,"syntology":null},{"paper":null,"slug":"one-nose-but-two-nostrils-learn-to-align-with","title":"One nose but two nostrils: Learn to align with sparse connections between two olfactory cortices","date":"2024-05-06","arxiv_id":"2405.03602","n_code_links":0,"syntology":null},{"paper":"/paper/the-privacy-power-of-correlated-noise-in","slug":"the-privacy-power-of-correlated-noise-in","title":"The Privacy Power of Correlated Noise in Decentralized Learning","date":"2024-05-02","arxiv_id":"2405.01031","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["elfirdoussilab1/DECOR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/hybrid-quantum-classical-scheduling-for","slug":"hybrid-quantum-classical-scheduling-for","title":"Q-Newton: Hybrid Quantum-Classical Scheduling for Accelerating Neural Network Training with Newton's Gradient Descent","date":"2024-04-30","arxiv_id":"2405.00252","n_code_links":1,"syntology":null},{"paper":null,"slug":"matching-the-statistical-query-lower-bound","title":"Matching the Statistical Query Lower Bound for $k$-Sparse Parity Problems with Sign Stochastic Gradient Descent","date":"2024-04-18","arxiv_id":"2404.12376","n_code_links":0,"syntology":null},{"paper":null,"slug":"singular-limit-analysis-of-gradient-descent","title":"Singular-limit analysis of gradient descent with noise injection","date":"2024-04-18","arxiv_id":"2404.12293","n_code_links":0,"syntology":null},{"paper":null,"slug":"clipped-sgd-algorithms-for-privacy-preserving","title":"Clipped SGD Algorithms for Performative Prediction: Tight Bounds for Clipping Bias and Remedies","date":"2024-04-17","arxiv_id":"2404.10995","n_code_links":0,"syntology":null},{"paper":null,"slug":"developing-lagrangian-based-methods-for","title":"Developing Lagrangian-based Methods for Nonsmooth Nonconvex Optimization","date":"2024-04-15","arxiv_id":"2404.09438","n_code_links":0,"syntology":null},{"paper":null,"slug":"proof-of-learning-with-incentive-security","title":"Proof-of-Learning with Incentive Security","date":"2024-04-13","arxiv_id":"2404.09005","n_code_links":0,"syntology":null},{"paper":"/paper/mope-mixture-of-prefix-experts-for-zero-shot","slug":"mope-mixture-of-prefix-experts-for-zero-shot","title":"MoPE: Mixture of Prefix Experts for Zero-Shot Dialogue State Tracking","date":"2024-04-12","arxiv_id":"2404.08559","n_code_links":1,"syntology":null},{"paper":null,"slug":"sliding-down-the-stairs-how-correlated-latent","title":"Sliding down the stairs: how correlated latent variables accelerate learning with neural networks","date":"2024-04-12","arxiv_id":"2404.08602","n_code_links":0,"syntology":null},{"paper":null,"slug":"variance-reduced-zeroth-order-methods-for","title":"Variance-reduced Zeroth-Order Methods for Fine-Tuning Language Models","date":"2024-04-11","arxiv_id":"2404.08080","n_code_links":0,"syntology":null},{"paper":"/paper/analysis-of-distributed-optimization","slug":"analysis-of-distributed-optimization","title":"PIM-Opt: Demystifying Distributed Optimization Algorithms on a Real-World Processing-In-Memory System","date":"2024-04-10","arxiv_id":"2404.07164","n_code_links":1,"syntology":null},{"paper":null,"slug":"simultaneous-linear-connectivity-of-neural","title":"Simultaneous linear connectivity of neural networks modulo permutation","date":"2024-04-09","arxiv_id":"2404.06498","n_code_links":0,"syntology":null},{"paper":"/paper/variational-stochastic-gradient-descent-for","slug":"variational-stochastic-gradient-descent-for","title":"Variational Stochastic Gradient Descent for Deep Neural Networks","date":"2024-04-09","arxiv_id":"2404.06549","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-continual-learning-with","title":"On the Convergence of Continual Learning with Adaptive Methods","date":"2024-04-08","arxiv_id":"2404.05555","n_code_links":0,"syntology":null},{"paper":null,"slug":"trustless-audits-without-revealing-data-or","title":"Trustless Audits without Revealing Data or Models","date":"2024-04-06","arxiv_id":"2404.04500","n_code_links":0,"syntology":null},{"paper":null,"slug":"faster-convergence-of-stochastic-accelerated","title":"Faster Convergence of Stochastic Accelerated Gradient Descent under Interpolation","date":"2024-04-03","arxiv_id":"2404.02378","n_code_links":0,"syntology":null},{"paper":"/paper/linear-combination-of-saved-checkpoints-makes","slug":"linear-combination-of-saved-checkpoints-makes","title":"Linear Combination of Saved Checkpoints Makes Consistency and Diffusion Models Better","date":"2024-04-02","arxiv_id":"2404.02241","n_code_links":1,"syntology":null},{"paper":"/paper/make-continual-learning-stronger-via-c-flat","slug":"make-continual-learning-stronger-via-c-flat","title":"Make Continual Learning Stronger via C-Flat","date":"2024-04-01","arxiv_id":"2404.00986","n_code_links":1,"syntology":{"ran":8,"of":12,"n_ran_checked":2,"n_instrument":6,"unverified":4,"pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","official":{"repos":["wannaa/c-flat"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"new-logarithmic-step-size-for-stochastic","title":"New logarithmic step size for stochastic gradient descent","date":"2024-04-01","arxiv_id":"2404.01257","n_code_links":0,"syntology":null},{"paper":"/paper/tablepuppet-a-generic-framework-for","slug":"tablepuppet-a-generic-framework-for","title":"TablePuppet: A Generic Framework for Relational Federated Learning","date":"2024-03-23","arxiv_id":"2403.15839","n_code_links":1,"syntology":null},{"paper":"/paper/differentially-private-next-token-prediction","slug":"differentially-private-next-token-prediction","title":"Differentially Private Next-Token Prediction of Large Language Models","date":"2024-03-22","arxiv_id":"2403.15638","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":0,"n_instrument":5,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["james-flemings/pmixed"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"analysing-heavy-tail-properties-of-stochastic","title":"Analysing heavy-tail properties of Stochastic Gradient Descent by means of Stochastic Recurrence Equations","date":"2024-03-20","arxiv_id":"2403.13868","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-aware-llm-based-safe-control-against","title":"Context-aware LLM-based Safe Control Against Latent Risks","date":"2024-03-18","arxiv_id":"2403.11863","n_code_links":0,"syntology":null}],"record_sha256":"38d5f830fc69c9a02e319889a095ef558f205ad7292c0f2a66ed37ea87567336","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}