{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/8","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":8,"pages_in_order":21,"rows_per_page":100,"rows":[701,800],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/7","next":"/method/sgd/papers/9","papers":[{"paper":null,"slug":"is-stochastic-gradient-descent-near-optimal","title":"Is Stochastic Gradient Descent Near Optimal?","date":"2022-09-18","arxiv_id":"2209.08627","n_code_links":0,"syntology":null},{"paper":null,"slug":"renyi-differential-privacy-of-propose-test","title":"Renyi Differential Privacy of Propose-Test-Release and Applications to Private and Robust Machine Learning","date":"2022-09-16","arxiv_id":"2209.07716","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficiency-ordering-of-stochastic-gradient","title":"Efficiency Ordering of Stochastic Gradient Descent","date":"2022-09-15","arxiv_id":"2209.07446","n_code_links":0,"syntology":null},{"paper":"/paper/random-initialisations-performing-above","slug":"random-initialisations-performing-above","title":"Random initialisations performing above chance and how to find them","date":"2022-09-15","arxiv_id":"2209.07509","n_code_links":1,"syntology":null},{"paper":"/paper/optimization-without-backpropagation","slug":"optimization-without-backpropagation","title":"Optimization without Backpropagation","date":"2022-09-13","arxiv_id":"2209.06302","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["gbelouze/forward-gradient"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"personalized-federated-learning-with-4","title":"Personalized Federated Learning with Communication Compression","date":"2022-09-12","arxiv_id":"2209.05148","n_code_links":0,"syntology":null},{"paper":null,"slug":"differentially-private-stochastic-gradient","title":"Differentially Private Stochastic Gradient Descent with Low-Noise","date":"2022-09-09","arxiv_id":"2209.04188","n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-outer-optimization-in-adversarial","title":"Revisiting Outer Optimization in Adversarial Training","date":"2022-09-02","arxiv_id":"2209.01199","n_code_links":0,"syntology":null},{"paper":null,"slug":"meta-objective-guided-disambiguation-for","title":"Meta Objective Guided Disambiguation for Partial Label Learning","date":"2022-08-26","arxiv_id":"2208.12459","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-simplified-convergence-theory-for-byzantine","title":"A simplified convergence theory for Byzantine resilient stochastic gradient descent","date":"2022-08-25","arxiv_id":"2208.11879","n_code_links":0,"syntology":null},{"paper":"/paper/accelerating-sgd-for-highly-ill-conditioned","slug":"accelerating-sgd-for-highly-ill-conditioned","title":"Accelerating SGD for Highly Ill-Conditioned Huge-Scale Online Matrix Completion","date":"2022-08-24","arxiv_id":"2208.11246","n_code_links":1,"syntology":null},{"paper":null,"slug":"robustness-to-unbounded-smoothness-of","title":"Robustness to Unbounded Smoothness of Generalized SignSGD","date":"2022-08-23","arxiv_id":"2208.11195","n_code_links":0,"syntology":null},{"paper":null,"slug":"provable-adaptivity-in-adam","title":"Provable Adaptivity of Adam under Non-uniform Smoothness","date":"2022-08-21","arxiv_id":"2208.09900","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-with-local-gradients-at-the-edge","title":"Learning with Local Gradients at the Edge","date":"2022-08-17","arxiv_id":"2208.08503","n_code_links":0,"syntology":null},{"paper":null,"slug":"training-overparametrized-neural-networks-in-1","title":"Training Overparametrized Neural Networks in Sublinear Time","date":"2022-08-09","arxiv_id":"2208.04508","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-stochastic-gradient-descent-for-fast","title":"Adaptive Stochastic Gradient Descent for Fast and Communication-Efficient Distributed Learning","date":"2022-08-04","arxiv_id":"2208.03134","n_code_links":0,"syntology":null},{"paper":null,"slug":"feature-selection-with-gradient-descent-on","title":"Feature selection with gradient descent on two-layer networks in low-rotation regimes","date":"2022-08-04","arxiv_id":"2208.02789","n_code_links":0,"syntology":null},{"paper":null,"slug":"formal-guarantees-for-heuristic-optimization","title":"Formal guarantees for heuristic optimization algorithms used in machine learning","date":"2022-07-31","arxiv_id":"2208.00502","n_code_links":0,"syntology":null},{"paper":"/paper/drsom-a-dimension-reduced-second-order-method","slug":"drsom-a-dimension-reduced-second-order-method","title":"DRSOM: A Dimension Reduced Second-Order Method","date":"2022-07-30","arxiv_id":"2208.00208","n_code_links":3,"syntology":null},{"paper":"/paper/cram-a-compression-aware-minimizer","slug":"cram-a-compression-aware-minimizer","title":"CrAM: A Compression-Aware Minimizer","date":"2022-07-28","arxiv_id":"2207.14200","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["ist-daslab/cram"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/dadao-decoupled-accelerated-decentralized","slug":"dadao-decoupled-accelerated-decentralized","title":"DADAO: Decoupled Accelerated Decentralized Asynchronous Optimization","date":"2022-07-26","arxiv_id":"2208.00779","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-step-size-methods-for-compressed-sgd","title":"Adaptive Step-Size Methods for Compressed SGD","date":"2022-07-20","arxiv_id":"2207.10046","n_code_links":0,"syntology":null},{"paper":"/paper/moment-centralization-based-gradient-descent","slug":"moment-centralization-based-gradient-descent","title":"Moment Centralization based Gradient Descent Optimizers for Convolutional Neural Networks","date":"2022-07-19","arxiv_id":"2207.09066","n_code_links":1,"syntology":null},{"paper":null,"slug":"hidden-progress-in-deep-learning-sgd-learns","title":"Hidden Progress in Deep Learning: SGD Learns Parities Near the Computational Limit","date":"2022-07-18","arxiv_id":"2207.08799","n_code_links":0,"syntology":null},{"paper":"/paper/communication-efficient-distributed-learning-5","slug":"communication-efficient-distributed-learning-5","title":"Communication-efficient Distributed Learning for Large Batch Optimization","date":"2022-07-17","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"sp2-a-second-order-stochastic-polyak-method","title":"SP2: A Second Order Stochastic Polyak Method","date":"2022-07-17","arxiv_id":"2207.08171","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-sketches-for-robust-regression-with","title":"Adaptive Sketches for Robust Regression with Importance Sampling","date":"2022-07-16","arxiv_id":"2207.07822","n_code_links":0,"syntology":null},{"paper":null,"slug":"mixtailor-mixed-gradient-aggregation-for","title":"MixTailor: Mixed Gradient Aggregation for Robust Learning Against Tailored Attacks","date":"2022-07-16","arxiv_id":"2207.07941","n_code_links":0,"syntology":null},{"paper":"/paper/tct-convexifying-federated-learning-using","slug":"tct-convexifying-federated-learning-using","title":"TCT: Convexifying Federated Learning using Bootstrapped Neural Tangent Kernels","date":"2022-07-13","arxiv_id":"2207.06343","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yaodongyu/tct"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-understanding-how-momentum-improves-1","title":"Towards understanding how momentum improves generalization in deep learning","date":"2022-07-13","arxiv_id":"2207.05931","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-stochastic-gradient-descent-and","title":"Stochastic Gradient Descent and Anomaly of Variance-flatness Relation in Artificial Neural Networks","date":"2022-07-11","arxiv_id":"2207.04932","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-uniform-in-time-diffusion-approximation","title":"On uniform-in-time diffusion approximation for stochastic gradient descent","date":"2022-07-11","arxiv_id":"2207.04922","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-the-number-of-tasks-in-continual","title":"Challenging Common Assumptions about Catastrophic Forgetting","date":"2022-07-10","arxiv_id":"2207.04543","n_code_links":0,"syntology":null},{"paper":null,"slug":"improved-binary-forward-exploration-learning","title":"Improved Binary Forward Exploration: Learning Rate Scheduling Method for Stochastic Optimization","date":"2022-07-09","arxiv_id":"2207.04198","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-poisson-binomial-mechanism-for-secure-and","title":"The Poisson binomial mechanism for secure and private federated learning","date":"2022-07-09","arxiv_id":"2207.09916","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deep-model-for-partial-multi-label-image","title":"A Deep Model for Partial Multi-Label Image Classification with Curriculum Based Disambiguation","date":"2022-07-06","arxiv_id":"2207.02410","n_code_links":0,"syntology":null},{"paper":null,"slug":"bfe-and-adabfe-a-new-approach-in-learning","title":"BFE and AdaBFE: A New Approach in Learning Rate Automation for Stochastic Optimization","date":"2022-07-06","arxiv_id":"2207.02763","n_code_links":0,"syntology":null},{"paper":null,"slug":"when-does-sgd-favor-flat-minima-a","title":"The alignment property of SGD noise and how it helps select flat minima: A stability analysis","date":"2022-07-06","arxiv_id":"2207.02628","n_code_links":0,"syntology":null},{"paper":null,"slug":"st-conal-consistency-based-acquisition","title":"ST-CoNAL: Consistency-Based Acquisition Criterion Using Temporal Self-Ensemble for Active Learning","date":"2022-07-05","arxiv_id":"2207.02182","n_code_links":0,"syntology":null},{"paper":"/paper/characterizing-the-effect-of-class-imbalance","slug":"characterizing-the-effect-of-class-imbalance","title":"A Theoretical Analysis of the Learning Dynamics under Class Imbalance","date":"2022-07-01","arxiv_id":"2207.00391","n_code_links":1,"syntology":null},{"paper":null,"slug":"studying-generalization-through-data","title":"Studying Generalization Through Data Averaging","date":"2022-06-28","arxiv_id":"2206.13669","n_code_links":0,"syntology":null},{"paper":null,"slug":"making-look-ahead-active-learning-strategies","title":"Making Look-Ahead Active Learning Strategies Feasible with Neural Tangent Kernels","date":"2022-06-25","arxiv_id":"2206.12569","n_code_links":0,"syntology":null},{"paper":null,"slug":"statistical-inference-with-implicit-sgd","title":"Statistical inference with implicit SGD: proximal Robbins-Monro vs. Polyak-Ruppert","date":"2022-06-25","arxiv_id":"2206.12663","n_code_links":0,"syntology":null},{"paper":"/paper/a-view-of-mini-batch-sgd-via-generating","slug":"a-view-of-mini-batch-sgd-via-generating","title":"A view of mini-batch SGD via generating functions: conditions of convergence, phase transitions, benefit from negative momenta","date":"2022-06-22","arxiv_id":"2206.11124","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["godofnothing/powerlawoptimization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-the-maximum-hessian-eigenvalue-and","title":"On the Maximum Hessian Eigenvalue and Generalization","date":"2022-06-21","arxiv_id":"2206.10654","n_code_links":0,"syntology":null},{"paper":"/paper/low-precision-stochastic-gradient-langevin-1","slug":"low-precision-stochastic-gradient-langevin-1","title":"Low-Precision Stochastic Gradient Langevin Dynamics","date":"2022-06-20","arxiv_id":"2206.09909","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ruqizhang/low-precision-sgld"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"provable-generalization-of-overparameterized","title":"Provable Generalization of Overparameterized Meta-learning Trained with SGD","date":"2022-06-18","arxiv_id":"2206.09136","n_code_links":0,"syntology":null},{"paper":"/paper/a-closer-look-at-smoothness-in-domain-1","slug":"a-closer-look-at-smoothness-in-domain-1","title":"A Closer Look at Smoothness in Domain Adversarial Training","date":"2022-06-16","arxiv_id":"2206.08213","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["val-iisc/sdat"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"sharper-convergence-guarantees-for","title":"Sharper Convergence Guarantees for Asynchronous SGD for Distributed and Federated Learning","date":"2022-06-16","arxiv_id":"2206.08307","n_code_links":0,"syntology":null},{"paper":"/paper/asynchronous-sgd-beats-minibatch-sgd-under","slug":"asynchronous-sgd-beats-minibatch-sgd-under","title":"Asynchronous SGD Beats Minibatch SGD Under Arbitrary Delays","date":"2022-06-15","arxiv_id":"2206.07638","n_code_links":1,"syntology":null},{"paper":null,"slug":"implicit-regularization-or-implicit","title":"Implicit Regularization or Implicit Conditioning? Exact Risk Trajectories of SGD in High Dimensions","date":"2022-06-15","arxiv_id":"2206.07252","n_code_links":0,"syntology":null},{"paper":"/paper/markov-chain-score-ascent-a-unifying","slug":"markov-chain-score-ascent-a-unifying","title":"Markov Chain Score Ascent: A Unifying Framework of Variational Inference with Markovian Gradients","date":"2022-06-13","arxiv_id":"2206.06295","n_code_links":1,"syntology":null},{"paper":null,"slug":"sgd-noise-and-implicit-low-rank-bias-in-deep","title":"SGD and Weight Decay Secretly Minimize the Rank of Your Neural Network","date":"2022-06-12","arxiv_id":"2206.05794","n_code_links":0,"syntology":null},{"paper":"/paper/stochastic-gradient-descent-without-full-data","slug":"stochastic-gradient-descent-without-full-data","title":"Stochastic Gradient Descent without Full Data Shuffle","date":"2022-06-12","arxiv_id":"2206.05830","n_code_links":1,"syntology":null},{"paper":"/paper/bayesian-estimation-of-differential-privacy","slug":"bayesian-estimation-of-differential-privacy","title":"Bayesian Estimation of Differential Privacy","date":"2022-06-10","arxiv_id":"2206.05199","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/responsible-ai-toolbox-privacy"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/trajectory-dependent-generalization-bounds","slug":"trajectory-dependent-generalization-bounds","title":"Learning Non-Vacuous Generalization Bounds from Optimization","date":"2022-06-09","arxiv_id":"2206.04359","n_code_links":1,"syntology":null},{"paper":null,"slug":"high-dimensional-limit-theorems-for-sgd","title":"High-dimensional limit theorems for SGD: Effective dynamics and critical scaling","date":"2022-06-08","arxiv_id":"2206.04030","n_code_links":0,"syntology":null},{"paper":null,"slug":"click-prediction-boosting-via-ensemble","title":"Click prediction boosting via Bayesian hyperparameter optimization based ensemble learning pipelines","date":"2022-06-07","arxiv_id":"2206.03592","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalization-error-bounds-for-deep-neural","title":"Generalization Error Bounds for Deep Neural Networks Trained by SGD","date":"2022-06-07","arxiv_id":"2206.03299","n_code_links":0,"syntology":null},{"paper":"/paper/integrating-random-effects-in-deep-neural","slug":"integrating-random-effects-in-deep-neural","title":"Integrating Random Effects in Deep Neural Networks","date":"2022-06-07","arxiv_id":"2206.03314","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gsimchoni/lmmnn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/early-stage-convergence-and-global","slug":"early-stage-convergence-and-global","title":"Early Stage Convergence and Global Convergence of Training Mildly Parameterized Neural Networks","date":"2022-06-05","arxiv_id":"2206.02139","n_code_links":1,"syntology":null},{"paper":"/paper/sharper-rates-and-flexible-framework-for","slug":"sharper-rates-and-flexible-framework-for","title":"Sharper Rates and Flexible Framework for Nonconvex SGD with Client and Data Sampling","date":"2022-06-05","arxiv_id":"2206.02275","n_code_links":1,"syntology":null},{"paper":"/paper/surprising-instabilities-in-training-deep","slug":"surprising-instabilities-in-training-deep","title":"A PDE-based Explanation of Extreme Numerical Sensitivities and Edge of Stability in Training Neural Networks","date":"2022-06-04","arxiv_id":"2206.02001","n_code_links":1,"syntology":null},{"paper":null,"slug":"algorithmic-stability-of-heavy-tailed","title":"Algorithmic Stability of Heavy-Tailed Stochastic Gradient Descent on Least Squares","date":"2022-06-02","arxiv_id":"2206.01274","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-descent-introduces-an","title":"Stochastic gradient descent introduces an effective landscape-dependent regularization favoring flat solutions","date":"2022-06-02","arxiv_id":"2206.01246","n_code_links":0,"syntology":null},{"paper":null,"slug":"trajectory-of-mini-batch-momentum-batch-size","title":"Trajectory of Mini-Batch Momentum: Batch Size Saturation and Convergence in High Dimensions","date":"2022-06-02","arxiv_id":"2206.01029","n_code_links":0,"syntology":null},{"paper":"/paper/a-theoretical-framework-for-inference","slug":"a-theoretical-framework-for-inference","title":"A Theoretical Framework for Inference Learning","date":"2022-06-01","arxiv_id":"2206.00164","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nalonso2/iltheory"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/computing-the-variance-of-shuffling","slug":"computing-the-variance-of-shuffling","title":"Computing the Variance of Shuffling Stochastic Gradient Algorithms via Power Spectral Density Analysis","date":"2022-06-01","arxiv_id":"2206.00632","n_code_links":1,"syntology":null},{"paper":"/paper/optimization-with-access-to-auxiliary","slug":"optimization-with-access-to-auxiliary","title":"Optimization with Access to Auxiliary Information","date":"2022-06-01","arxiv_id":"2206.00395","n_code_links":1,"syntology":null},{"paper":"/paper/metrizing-fairness","slug":"metrizing-fairness","title":"Metrizing Fairness","date":"2022-05-30","arxiv_id":"2205.15049","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-avoiding-local-minima-using-gradient","title":"Special Properties of Gradient Descent with Large Learning Rates","date":"2022-05-30","arxiv_id":"2205.15142","n_code_links":0,"syntology":null},{"paper":"/paper/re-parameterizing-your-optimizers-rather-than","slug":"re-parameterizing-your-optimizers-rather-than","title":"Re-parameterizing Your Optimizers rather than Architectures","date":"2022-05-30","arxiv_id":"2205.15242","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["dingxiaoh/repoptimizers"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generalization-bounds-for-gradient-methods","title":"Generalization Bounds for Gradient Methods via Discrete and Continuous Prior","date":"2022-05-27","arxiv_id":"2205.13799","n_code_links":0,"syntology":null},{"paper":null,"slug":"privacy-of-noisy-stochastic-gradient-descent","title":"Privacy of Noisy Stochastic Gradient Descent: More Iterations without More Privacy Loss","date":"2022-05-27","arxiv_id":"2205.13710","n_code_links":0,"syntology":null},{"paper":"/paper/fedaug-reducing-the-local-learning-bias","slug":"fedaug-reducing-the-local-learning-bias","title":"FedBR: Improving Federated Learning on Heterogeneous Data via Local Learning Bias Reduction","date":"2022-05-26","arxiv_id":"2205.13462","n_code_links":1,"syntology":{"ran":10,"of":15,"n_ran_checked":6,"n_instrument":4,"unverified":5,"pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","official":{"repos":["lins-lab/fedbr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/trainable-weight-averaging-for-fast","slug":"trainable-weight-averaging-for-fast","title":"Trainable Weight Averaging: A General Approach for Subspace Training","date":"2022-05-26","arxiv_id":"2205.13104","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 5 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["nblt/twa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mirror-descent-maximizes-generalized-margin","title":"Mirror Descent Maximizes Generalized Margin and Can Be Implemented Efficiently","date":"2022-05-25","arxiv_id":"2205.12808","n_code_links":0,"syntology":null},{"paper":null,"slug":"byzantine-machine-learning-made-easy-by","title":"Byzantine Machine Learning Made Easy by Resilient Averaging of Momentums","date":"2022-05-24","arxiv_id":"2205.12173","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-deep-learning-models-and-news-headlines","title":"Do Deep Learning Models and News Headlines Outperform Conventional Prediction Techniques on Forex Data?","date":"2022-05-22","arxiv_id":"2205.10743","n_code_links":0,"syntology":null},{"paper":"/paper/grab-finding-provably-better-data","slug":"grab-finding-provably-better-data","title":"GraB: Finding Provably Better Data Permutations than Random Reshuffling","date":"2022-05-22","arxiv_id":"2205.10733","n_code_links":3,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["eugenelyc/grab"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/on-the-sdes-and-scaling-rules-for-adaptive","slug":"on-the-sdes-and-scaling-rules-for-adaptive","title":"On the SDEs and Scaling Rules for Adaptive Gradient Algorithms","date":"2022-05-20","arxiv_id":"2205.10287","n_code_links":1,"syntology":null},{"paper":null,"slug":"homogenization-of-sgd-in-high-dimensions","title":"Homogenization of SGD in high-dimensions: Exact dynamics and generalization properties","date":"2022-05-14","arxiv_id":"2205.07069","n_code_links":0,"syntology":null},{"paper":null,"slug":"heavy-tail-phenomenon-in-decentralized-sgd","title":"Heavy-Tail Phenomenon in Decentralized SGD","date":"2022-05-13","arxiv_id":"2205.06689","n_code_links":0,"syntology":null},{"paper":"/paper/a-communication-efficient-distributed-3","slug":"a-communication-efficient-distributed-3","title":"A Communication-Efficient Distributed Gradient Clipping Algorithm for Training Deep Neural Networks","date":"2022-05-10","arxiv_id":"2205.05040","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mingruiliu-ml-lab/communication-efficient-local-gradient-clipping"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"unsupervised-slot-schema-induction-for-task-1","title":"Unsupervised Slot Schema Induction for Task-oriented Dialog","date":"2022-05-09","arxiv_id":"2205.04515","n_code_links":0,"syntology":null},{"paper":null,"slug":"random-reshuffling-with-variance-reduction-1","title":"Federated Random Reshuffling with Compression and Variance Reduction","date":"2022-05-08","arxiv_id":"2205.03914","n_code_links":0,"syntology":null},{"paper":null,"slug":"making-sgd-parameter-free","title":"Making SGD Parameter-Free","date":"2022-05-04","arxiv_id":"2205.02160","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-directional-bias-helps-stochastic","title":"The Directional Bias Helps Stochastic Gradient Descent to Generalize in Kernel Regression Models","date":"2022-04-29","arxiv_id":"2205.00061","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-lipschitz-sharp-generalization-and","title":"Beyond Lipschitz: Sharp Generalization and Excess Risk Bounds for Full-Batch GD","date":"2022-04-26","arxiv_id":"2204.12446","n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-stochastic-primal-dual-learning","title":"Federated Stochastic Primal-dual Learning with Differential Privacy","date":"2022-04-26","arxiv_id":"2204.12284","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-acceleration-of-gradient-based-empirical","title":"On Acceleration of Gradient-Based Empirical Risk Minimization using Local Polynomial Regression","date":"2022-04-16","arxiv_id":"2204.07702","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-schema-graph-fusion-network-for-multi-1","title":"Dynamic Schema Graph Fusion Network for Multi-Domain Dialogue State Tracking","date":"2022-04-14","arxiv_id":"2204.06677","n_code_links":0,"syntology":null},{"paper":"/paper/what-you-see-is-what-you-get-distributional","slug":"what-you-see-is-what-you-get-distributional","title":"What You See is What You Get: Principled Deep Learning via Distributional Generalization","date":"2022-04-07","arxiv_id":"2204.03230","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":4,"n_instrument":2,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yangarbiter/dp-dg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"nonlinear-gradient-mappings-and-stochastic","title":"Nonlinear gradient mappings and stochastic optimization: A general framework with applications to heavy-tail noise","date":"2022-04-06","arxiv_id":"2204.02593","n_code_links":0,"syntology":null},{"paper":null,"slug":"privacy-preserving-federated-learning-via","title":"Privacy-Preserving Federated Learning via System Immersion and Random Matrix Encryption","date":"2022-04-05","arxiv_id":"2204.02497","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-stochastic-gradient-descent-and","title":"Deep learning, stochastic gradient descent and diffusion maps","date":"2022-04-04","arxiv_id":"2204.01365","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-sampling-affects-the-complexity-of","title":"Data Sampling Affects the Complexity of Online SGD over Dependent Data","date":"2022-03-31","arxiv_id":"2204.00006","n_code_links":0,"syntology":null},{"paper":"/paper/exploiting-explainable-metrics-for-augmented","slug":"exploiting-explainable-metrics-for-augmented","title":"Exploiting Explainable Metrics for Augmented SGD","date":"2022-03-31","arxiv_id":"2203.16723","n_code_links":2,"syntology":null},{"paper":"/paper/conjugate-gradient-method-for-generative","slug":"conjugate-gradient-method-for-generative","title":"Conjugate Gradient Method for Generative Adversarial Networks","date":"2022-03-28","arxiv_id":"2203.14495","n_code_links":1,"syntology":null},{"paper":"/paper/a-robust-optimization-method-for-label-noisy","slug":"a-robust-optimization-method-for-label-noisy","title":"A Robust Optimization Method for Label Noisy Datasets Based on Adaptive Threshold: Adaptive-k","date":"2022-03-26","arxiv_id":"2203.14165","n_code_links":1,"syntology":null}],"record_sha256":"f9f9b4ec4964897a2c45f20b4971d046c0bf0a2c028694c306bf952f0894fe44","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}