{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/7","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":7,"pages_in_order":21,"rows_per_page":100,"rows":[601,700],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/6","next":"/method/sgd/papers/8","papers":[{"paper":null,"slug":"the-monge-gap-a-regularizer-to-learn-all","title":"The Monge Gap: A Regularizer to Learn All Transport Maps","date":"2023-02-09","arxiv_id":"2302.04953","n_code_links":0,"syntology":null},{"paper":"/paper/dog-is-sgd-s-best-friend-a-parameter-free","slug":"dog-is-sgd-s-best-friend-a-parameter-free","title":"DoG is SGD's Best Friend: A Parameter-Free Dynamic Step Size Schedule","date":"2023-02-08","arxiv_id":"2302.12022","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["formll/dog"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/easy-learning-from-label-proportions","slug":"easy-learning-from-label-proportions","title":"Easy Learning from Label Proportions","date":"2023-02-06","arxiv_id":"2302.03115","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-federated-averaging","title":"On the Convergence of Federated Averaging with Cyclic Client Participation","date":"2023-02-06","arxiv_id":"2302.03109","n_code_links":0,"syntology":null},{"paper":"/paper/toward-large-kernel-models","slug":"toward-large-kernel-models","title":"Toward Large Kernel Models","date":"2023-02-06","arxiv_id":"2302.02605","n_code_links":1,"syntology":null},{"paper":null,"slug":"z-signfedavg-a-unified-stochastic-sign-based","title":"$z$-SignFedAvg: A Unified Stochastic Sign-based Compression for Federated Learning","date":"2023-02-06","arxiv_id":"2302.02589","n_code_links":0,"syntology":null},{"paper":null,"slug":"quantized-distributed-training-of-large","title":"Quantized Distributed Training of Large Models with Convergence Guarantees","date":"2023-02-05","arxiv_id":"2302.02390","n_code_links":0,"syntology":null},{"paper":null,"slug":"selecting-the-best-optimizers-for-deep","title":"Selecting the Best Optimizers for Deep Learning based Medical Image Segmentation","date":"2023-02-05","arxiv_id":"2302.02289","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-suppressing-range-of-adaptive-stepsizes-of","title":"On Suppressing Range of Adaptive Stepsizes of Adam to Improve Generalisation Performance","date":"2023-02-02","arxiv_id":"2302.01029","n_code_links":0,"syntology":null},{"paper":"/paper/scale-up-with-order-finding-good-data","slug":"scale-up-with-order-finding-good-data","title":"Coordinating Distributed Example Orders for Provably Accelerated Training","date":"2023-02-02","arxiv_id":"2302.00845","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["garlguo/cd-grab"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/step-learning-n-m-structured-sparsity-masks","slug":"step-learning-n-m-structured-sparsity-masks","title":"STEP: Learning N:M Structured Sparsity Masks from Scratch with Precondition","date":"2023-02-02","arxiv_id":"2302.01172","n_code_links":1,"syntology":null},{"paper":null,"slug":"dissecting-the-effects-of-sgd-noise-in","title":"Dissecting the Effects of SGD Noise in Distinct Regimes of Deep Learning","date":"2023-01-31","arxiv_id":"2301.13703","n_code_links":0,"syntology":null},{"paper":"/paper/schema-guided-semantic-accuracy-faithfulness","slug":"schema-guided-semantic-accuracy-faithfulness","title":"Schema-Guided Semantic Accuracy: Faithfulness in Task-Oriented Dialogue Response Generation","date":"2023-01-29","arxiv_id":"2301.12568","n_code_links":1,"syntology":null},{"paper":null,"slug":"cyclicfl-a-cyclic-model-pre-training-approach","title":"CyclicFL: A Cyclic Model Pre-Training Approach to Efficient Federated Learning","date":"2023-01-28","arxiv_id":"2301.12193","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-lipschitz-constant-of-deep-networks","slug":"on-the-lipschitz-constant-of-deep-networks","title":"On the Lipschitz Constant of Deep Networks and Double Descent","date":"2023-01-28","arxiv_id":"2301.12309","n_code_links":1,"syntology":null},{"paper":null,"slug":"algorithmic-stability-of-heavy-tailed-sgd","title":"Algorithmic Stability of Heavy-Tailed SGD with General Loss Functions","date":"2023-01-27","arxiv_id":"2301.11885","n_code_links":0,"syntology":null},{"paper":"/paper/trainable-activations-for-image","slug":"trainable-activations-for-image","title":"Trainable Activations for Image Classification","date":"2023-01-26","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/transfer-learning-in-deep-learning-models-for","slug":"transfer-learning-in-deep-learning-models-for","title":"Transfer Learning in Deep Learning Models for Building Load Forecasting: Case of Limited Data","date":"2023-01-25","arxiv_id":"2301.10663","n_code_links":2,"syntology":null},{"paper":"/paper/read-the-signs-towards-invariance-to-gradient","slug":"read-the-signs-towards-invariance-to-gradient","title":"Read the Signs: Towards Invariance to Gradient Descent's Hyperparameter Initialization","date":"2023-01-24","arxiv_id":"2301.10133","n_code_links":1,"syntology":null},{"paper":null,"slug":"genetically-modified-wolf-optimization-with","title":"Genetically Modified Wolf Optimization with Stochastic Gradient Descent for Optimising Deep Neural Networks","date":"2023-01-21","arxiv_id":"2301.08950","n_code_links":0,"syntology":null},{"paper":"/paper/scadles-scalable-deep-learning-over-streaming","slug":"scadles-scalable-deep-learning-over-streaming","title":"ScaDLES: Scalable Deep Learning over Streaming data at the Edge","date":"2023-01-21","arxiv_id":"2301.08897","n_code_links":1,"syntology":null},{"paper":"/paper/learning-rate-free-learning-by-d-adaptation","slug":"learning-rate-free-learning-by-d-adaptation","title":"Learning-Rate-Free Learning by D-Adaptation","date":"2023-01-18","arxiv_id":"2301.07733","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["facebookresearch/dadaptation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"convergence-of-first-order-algorithms-for","title":"Convergence of First-Order Algorithms for Meta-Learning with Moreau Envelopes","date":"2023-01-17","arxiv_id":"2301.06806","n_code_links":0,"syntology":null},{"paper":"/paper/expected-gradients-of-maxout-networks-and","slug":"expected-gradients-of-maxout-networks-and","title":"Expected Gradients of Maxout Networks and Consequences to Parameter Initialization","date":"2023-01-17","arxiv_id":"2301.06956","n_code_links":1,"syntology":null},{"paper":null,"slug":"cedas-a-compressed-decentralized-stochastic","title":"CEDAS: A Compressed Decentralized Stochastic Gradient Method with Improved Convergence","date":"2023-01-14","arxiv_id":"2301.05872","n_code_links":0,"syntology":null},{"paper":null,"slug":"toward-theoretical-guidance-for-two-common","title":"Toward Theoretical Guidance for Two Common Questions in Practical Cross-Validation based Hyperparameter Selection","date":"2023-01-12","arxiv_id":"2301.05131","n_code_links":0,"syntology":null},{"paper":null,"slug":"training-trajectories-mini-batch-losses-and","title":"Training trajectories, mini-batch losses and the curious role of the learning rate","date":"2023-01-05","arxiv_id":"2301.02312","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-stochastic-gradient-2","title":"On the Convergence of Stochastic Gradient Descent in Low-precision Number Formats","date":"2023-01-04","arxiv_id":"2301.01651","n_code_links":0,"syntology":null},{"paper":null,"slug":"temporal-difference-learning-with-compressed","title":"Temporal Difference Learning with Compressed Updates: Error-Feedback meets Reinforcement Learning","date":"2023-01-03","arxiv_id":"2301.00944","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpu-accelerated-matrix-factorization-of-large","title":"GPU accelerated matrix factorization of large scale data using block based approach","date":"2023-01-02","arxiv_id":"2304.13724","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-statistical-inference-for-contextual","title":"Online Statistical Inference for Contextual Bandits via Stochastic Gradient Descent","date":"2022-12-30","arxiv_id":"2212.14883","n_code_links":0,"syntology":null},{"paper":null,"slug":"visualizing-information-bottleneck-through","title":"Visualizing Information Bottleneck through Variational Inference","date":"2022-12-24","arxiv_id":"2212.12667","n_code_links":0,"syntology":null},{"paper":"/paper/when-do-curricula-work-in-federated-learning","slug":"when-do-curricula-work-in-federated-learning","title":"When Do Curricula Work in Federated Learning?","date":"2022-12-24","arxiv_id":"2212.12712","n_code_links":1,"syntology":null},{"paper":null,"slug":"anytod-a-programmable-task-oriented-dialog","title":"AnyTOD: A Programmable Task-Oriented Dialog System","date":"2022-12-20","arxiv_id":"2212.09939","n_code_links":0,"syntology":null},{"paper":"/paper/improving-levenberg-marquardt-algorithm-for","slug":"improving-levenberg-marquardt-algorithm-for","title":"Improving Levenberg-Marquardt Algorithm for Neural Networks","date":"2022-12-17","arxiv_id":"2212.08769","n_code_links":1,"syntology":null},{"paper":null,"slug":"huber-energy-measure-quantization","title":"Huber-energy measure quantization","date":"2022-12-15","arxiv_id":"2212.08162","n_code_links":0,"syntology":null},{"paper":null,"slug":"jax-accelerated-neuroevolution-of-physics","title":"Neuroevolution of Physics-Informed Neural Nets: Benchmark Problems and Comparative Results","date":"2022-12-15","arxiv_id":"2212.07624","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalizing-dp-sgd-with-shuffling-and","title":"Generalizing DP-SGD with Shuffling and Batch Clipping","date":"2022-12-12","arxiv_id":"2212.05796","n_code_links":0,"syntology":null},{"paper":null,"slug":"covariance-estimators-for-the-root-sgd","title":"Covariance Estimators for the ROOT-SGD Algorithm in Online Learning","date":"2022-12-02","arxiv_id":"2212.01259","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-certain-choices-of-cnn","title":"Investigating certain choices of CNN configurations for brain lesion segmentation","date":"2022-12-02","arxiv_id":"2212.01235","n_code_links":0,"syntology":null},{"paper":"/paper/disentangling-the-mechanisms-behind-implicit","slug":"disentangling-the-mechanisms-behind-implicit","title":"Disentangling the Mechanisms Behind Implicit Regularization in SGD","date":"2022-11-29","arxiv_id":"2211.15853","n_code_links":1,"syntology":null},{"paper":null,"slug":"stochastic-steffensen-method","title":"Stochastic Steffensen method","date":"2022-11-28","arxiv_id":"2211.15310","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-compact-features-via-in-training","title":"Learning Compact Features via In-Training Representation Alignment","date":"2022-11-23","arxiv_id":"2211.13332","n_code_links":0,"syntology":null},{"paper":null,"slug":"mutual-information-learned-regressor-an","title":"Mutual Information Learned Regressor: an Information-theoretic Viewpoint of Training Regression Systems","date":"2022-11-23","arxiv_id":"2211.12685","n_code_links":0,"syntology":null},{"paper":"/paper/modeldiff-a-framework-for-comparing-learning","slug":"modeldiff-a-framework-for-comparing-learning","title":"ModelDiff: A Framework for Comparing Learning Algorithms","date":"2022-11-22","arxiv_id":"2211.12491","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["madrylab/modeldiff"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/scaling-up-dataset-distillation-to-imagenet","slug":"scaling-up-dataset-distillation-to-imagenet","title":"Scaling Up Dataset Distillation to ImageNet-1K with Constant Memory","date":"2022-11-19","arxiv_id":"2211.10586","n_code_links":2,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["justincui03/tesla"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"two-facets-of-sde-under-an-information","title":"Two Facets of SDE Under an Information-Theoretic Lens: Generalization of SGD via Training Trajectories and via Terminal States","date":"2022-11-19","arxiv_id":"2211.10691","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-to-fine-tune-vision-models-with-sgd","title":"How to Fine-Tune Vision Models with SGD","date":"2022-11-17","arxiv_id":"2211.09359","n_code_links":0,"syntology":null},{"paper":"/paper/sketchysgd-reliable-stochastic-optimization","slug":"sketchysgd-reliable-stochastic-optimization","title":"SketchySGD: Reliable Stochastic Optimization via Randomized Curvature Estimates","date":"2022-11-16","arxiv_id":"2211.08597","n_code_links":1,"syntology":null},{"paper":"/paper/empirical-study-on-optimizer-selection-for","slug":"empirical-study-on-optimizer-selection-for","title":"Empirical Study on Optimizer Selection for Out-of-Distribution Generalization","date":"2022-11-15","arxiv_id":"2211.08583","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":3,"n_instrument":4,"unverified":4,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","official":{"repos":["hiroki11x/optimizer_comparison_ood"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/repair-renormalizing-permuted-activations-for","slug":"repair-renormalizing-permuted-activations-for","title":"REPAIR: REnormalizing Permuted Activations for Interpolation Repair","date":"2022-11-15","arxiv_id":"2211.08403","n_code_links":1,"syntology":null},{"paper":null,"slug":"selective-memory-recursive-least-squares","title":"Selective Memory Recursive Least Squares: Recast Forgetting into Memory in RBF Neural Network Based Real-Time Learning","date":"2022-11-15","arxiv_id":"2211.07909","n_code_links":0,"syntology":null},{"paper":"/paper/alternating-implicit-projected-sgd-and-its","slug":"alternating-implicit-projected-sgd-and-its","title":"Alternating Implicit Projected SGD and Its Efficient Variants for Equality-constrained Bilevel Optimization","date":"2022-11-14","arxiv_id":"2211.07096","n_code_links":1,"syntology":null},{"paper":"/paper/multi-epoch-matrix-factorization-mechanisms","slug":"multi-epoch-matrix-factorization-mechanisms","title":"Multi-Epoch Matrix Factorization Mechanisms for Private Machine Learning","date":"2022-11-12","arxiv_id":"2211.06530","n_code_links":1,"syntology":null},{"paper":null,"slug":"variants-of-sgd-for-lipschitz-continuous-loss","title":"Variants of SGD for Lipschitz Continuous Loss Functions in Low-Precision Environments","date":"2022-11-09","arxiv_id":"2211.04655","n_code_links":0,"syntology":null},{"paper":null,"slug":"askewsgd-an-annealed-interval-constrained","title":"AskewSGD : An Annealed interval-constrained Optimisation method to train Quantized Neural Networks","date":"2022-11-07","arxiv_id":"2211.03741","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-parallel-stochastic-gradient","title":"Accelerating Parallel Stochastic Gradient Descent via Non-blocking Mini-batches","date":"2022-11-02","arxiv_id":"2211.00889","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-deviations-rates-for-stochastic","title":"Large deviations rates for stochastic gradient descent with strongly convex functions","date":"2022-11-02","arxiv_id":"2211.00969","n_code_links":0,"syntology":null},{"paper":null,"slug":"rcd-sgd-resource-constrained-distributed-sgd","title":"RCD-SGD: Resource-Constrained Distributed SGD in Heterogeneous Environment via Submodular Partitioning","date":"2022-11-02","arxiv_id":"2211.00839","n_code_links":0,"syntology":null},{"paper":null,"slug":"strong-lottery-ticket-hypothesis-with","title":"Strong Lottery Ticket Hypothesis with $\\varepsilon$--perturbation","date":"2022-10-29","arxiv_id":"2210.16589","n_code_links":0,"syntology":null},{"paper":null,"slug":"flatter-faster-scaling-momentum-for-optimal","title":"Flatter, faster: scaling momentum for optimal speedup of SGD","date":"2022-10-28","arxiv_id":"2210.16400","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-mirror-descent-in-average-ensemble","title":"Stochastic Mirror Descent in Average Ensemble Models","date":"2022-10-27","arxiv_id":"2210.15323","n_code_links":0,"syntology":null},{"paper":null,"slug":"same-pre-training-loss-better-downstream","title":"Same Pre-training Loss, Better Downstream: Implicit Bias Matters for Language Models","date":"2022-10-25","arxiv_id":"2210.14199","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-top-k-in-sgd-for-communication","title":"Adaptive Top-K in SGD for Communication-Efficient Distributed Learning","date":"2022-10-24","arxiv_id":"2210.13532","n_code_links":0,"syntology":null},{"paper":null,"slug":"decentralized-stochastic-bilevel-optimization","title":"Decentralized Stochastic Bilevel Optimization with Improved per-Iteration Complexity","date":"2022-10-23","arxiv_id":"2210.12839","n_code_links":0,"syntology":null},{"paper":null,"slug":"k-sam-sharpness-aware-minimization-at-the","title":"K-SAM: Sharpness-Aware Minimization at the Speed of SGD","date":"2022-10-23","arxiv_id":"2210.12864","n_code_links":0,"syntology":null},{"paper":null,"slug":"when-expressivity-meets-trainability-fewer-1","title":"When Expressivity Meets Trainability: Fewer than $n$ Neurons Can Work","date":"2022-10-21","arxiv_id":"2210.12001","n_code_links":0,"syntology":null},{"paper":null,"slug":"global-convergence-of-sgd-on-two-layer-neural","title":"Global Convergence of SGD On Two Layer Neural Nets","date":"2022-10-20","arxiv_id":"2210.11452","n_code_links":0,"syntology":null},{"paper":"/paper/large-batch-optimization-for-dense-visual","slug":"large-batch-optimization-for-dense-visual","title":"Large-batch Optimization for Dense Visual Predictions","date":"2022-10-20","arxiv_id":"2210.11078","n_code_links":1,"syntology":null},{"paper":null,"slug":"local-sgd-in-overparameterized-linear","title":"Local SGD in Overparameterized Linear Regression","date":"2022-10-20","arxiv_id":"2210.11562","n_code_links":0,"syntology":null},{"paper":"/paper/dpis-an-enhanced-mechanism-for-differentially","slug":"dpis-an-enhanced-mechanism-for-differentially","title":"DPIS: An Enhanced Mechanism for Differentially Private SGD with Importance Sampling","date":"2022-10-18","arxiv_id":"2210.09634","n_code_links":1,"syntology":null},{"paper":null,"slug":"linear-scalarization-for-byzantine-robust","title":"Linear Scalarization for Byzantine-robust learning on non-IID data","date":"2022-10-15","arxiv_id":"2210.08287","n_code_links":0,"syntology":null},{"paper":"/paper/communication-efficient-topologies-for","slug":"communication-efficient-topologies-for","title":"Communication-Efficient Topologies for Decentralized Learning with $O(1)$ Consensus Rate","date":"2022-10-14","arxiv_id":"2210.07881","n_code_links":1,"syntology":null},{"paper":"/paper/when-adversarial-training-meets-vision","slug":"when-adversarial-training-meets-vision","title":"When Adversarial Training Meets Vision Transformers: Recipes from Training to Architecture","date":"2022-10-14","arxiv_id":"2210.07540","n_code_links":1,"syntology":null},{"paper":null,"slug":"from-gradient-flow-on-population-loss-to","title":"From Gradient Flow on Population Loss to Learning with Stochastic Gradient Descent","date":"2022-10-13","arxiv_id":"2210.06705","n_code_links":0,"syntology":null},{"paper":null,"slug":"mean-field-analysis-for-heavy-ball-methods","title":"Mean-field analysis for heavy ball methods: Dropout-stability, connectivity, and global convergence","date":"2022-10-13","arxiv_id":"2210.06819","n_code_links":0,"syntology":null},{"paper":"/paper/wasserstein-barycenter-based-model-fusion-and","slug":"wasserstein-barycenter-based-model-fusion-and","title":"Wasserstein Barycenter-based Model Fusion and Linear Mode Connectivity of Neural Networks","date":"2022-10-13","arxiv_id":"2210.06671","n_code_links":1,"syntology":null},{"paper":"/paper/adanorm-adaptive-gradient-norm-correction","slug":"adanorm-adaptive-gradient-norm-correction","title":"AdaNorm: Adaptive Gradient Norm Correction based Optimizer for CNNs","date":"2022-10-12","arxiv_id":"2210.06364","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-information-retention-in-large","title":"Improving information retention in large scale online continual learning","date":"2022-10-12","arxiv_id":"2210.06401","n_code_links":0,"syntology":null},{"paper":"/paper/rigorous-dynamical-mean-field-theory-for","slug":"rigorous-dynamical-mean-field-theory-for","title":"Rigorous dynamical mean field theory for stochastic gradient descent methods","date":"2022-10-12","arxiv_id":"2210.06591","n_code_links":1,"syntology":null},{"paper":"/paper/a-kernel-based-view-of-language-model-fine","slug":"a-kernel-based-view-of-language-model-fine","title":"A Kernel-Based View of Language Model Fine-Tuning","date":"2022-10-11","arxiv_id":"2210.05643","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["princeton-nlp/lm-kernel-ft"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/sgd-with-large-step-sizes-learns-sparse","slug":"sgd-with-large-step-sizes-learns-sparse","title":"SGD with Large Step Sizes Learns Sparse Features","date":"2022-10-11","arxiv_id":"2210.05337","n_code_links":1,"syntology":null},{"paper":null,"slug":"fast-hierarchical-learning-for-few-shot","title":"Fast Hierarchical Learning for Few-Shot Object Detection","date":"2022-10-10","arxiv_id":"2210.05008","n_code_links":0,"syntology":null},{"paper":null,"slug":"dissecting-adaptive-methods-in-gans","title":"Dissecting adaptive methods in GANs","date":"2022-10-09","arxiv_id":"2210.04319","n_code_links":0,"syntology":null},{"paper":null,"slug":"hetsyn-speeding-up-local-sgd-with","title":"STSyn: Speeding Up Local SGD with Straggler-Tolerant Synchronization","date":"2022-10-06","arxiv_id":"2210.03521","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-up-stochastic-gradient-descent-for","title":"Scaling up Stochastic Gradient Descent for Non-convex Optimisation","date":"2022-10-06","arxiv_id":"2210.02882","n_code_links":0,"syntology":null},{"paper":null,"slug":"unmasking-the-lottery-ticket-hypothesis-what","title":"Unmasking the Lottery Ticket Hypothesis: What's Encoded in a Winning Ticket's Mask?","date":"2022-10-06","arxiv_id":"2210.03044","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-an-invertible-output-mapping-can","title":"Learning an Invertible Output Mapping Can Mitigate Simplicity Bias in Neural Networks","date":"2022-10-04","arxiv_id":"2210.01360","n_code_links":0,"syntology":null},{"paper":null,"slug":"b-stochastic-sign-sgd-a-byzantine-resilient","title":"Distributed Non-Convex Optimization with One-Bit Compressors on Heterogeneous Data: Efficient and Resilient Algorithms","date":"2022-10-03","arxiv_id":"2210.00665","n_code_links":0,"syntology":null},{"paper":null,"slug":"downlink-compression-improves-topk","title":"Downlink Compression Improves TopK Sparsification","date":"2022-09-30","arxiv_id":"2209.15203","n_code_links":0,"syntology":null},{"paper":null,"slug":"momentum-tracking-momentum-acceleration-for","title":"Momentum Tracking: Momentum Acceleration for Decentralized Deep Learning on Heterogeneous Data","date":"2022-09-30","arxiv_id":"2209.15505","n_code_links":0,"syntology":null},{"paper":"/paper/nag-gs-semi-implicit-accelerated-and-robust","slug":"nag-gs-semi-implicit-accelerated-and-robust","title":"NAG-GS: Semi-Implicit, Accelerated and Robust Stochastic Optimizer","date":"2022-09-29","arxiv_id":"2209.14937","n_code_links":2,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["skolai/nag-gs","naggsopt/naggs"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"neural-networks-efficiently-learn-low","title":"Neural Networks Efficiently Learn Low-Dimensional Representations with SGD","date":"2022-09-29","arxiv_id":"2209.14863","n_code_links":0,"syntology":null},{"paper":null,"slug":"statistical-learning-and-inverse-problems-an","title":"Statistical Learning and Inverse Problems: A Stochastic Gradient Approach","date":"2022-09-29","arxiv_id":"2209.14967","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-stability-analysis-of-open-federated","title":"On the Stability Analysis of Open Federated Learning Systems","date":"2022-09-25","arxiv_id":"2209.12307","n_code_links":0,"syntology":null},{"paper":null,"slug":"error-mitigation-aided-optimization-of","title":"Error Mitigation-Aided Optimization of Parameterized Quantum Circuits: Convergence Analysis","date":"2022-09-23","arxiv_id":"2209.11514","n_code_links":0,"syntology":null},{"paper":"/paper/making-byzantine-decentralized-learning","slug":"making-byzantine-decentralized-learning","title":"Robust Collaborative Learning with Linear Gradient Overhead","date":"2022-09-22","arxiv_id":"2209.10931","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["lpd-epfl/robust-collaborative-learning"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"generalization-bounds-for-stochastic-gradient","title":"Generalization Bounds for Stochastic Gradient Descent via Localized $\\varepsilon$-Covers","date":"2022-09-19","arxiv_id":"2209.08951","n_code_links":0,"syntology":null},{"paper":null,"slug":"stability-and-generalization-analysis-of","title":"Stability and Generalization Analysis of Gradient Methods for Shallow Neural Networks","date":"2022-09-19","arxiv_id":"2209.09298","n_code_links":0,"syntology":null},{"paper":null,"slug":"empirical-analysis-on-top-k-gradient","title":"Empirical Analysis on Top-k Gradient Sparsification for Distributed Deep Learning in a Supercomputing Environment","date":"2022-09-18","arxiv_id":"2209.08497","n_code_links":0,"syntology":null}],"record_sha256":"c1c82a1a815a67734595d45fac83e3ad84f45d1f63ac537884cb10037745c6ed","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}