{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/15","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":15,"pages_in_order":21,"rows_per_page":100,"rows":[1401,1500],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/14","next":"/method/sgd/papers/16","papers":[{"paper":"/paper/understanding-the-difficulty-of-training","slug":"understanding-the-difficulty-of-training","title":"Understanding the Difficulty of Training Transformers","date":"2020-04-17","arxiv_id":"2004.08249","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":3,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["LiyuanLucasLiu/Transforemr-Clinic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"on-learning-rates-and-schrodinger-operators","title":"On Learning Rates and Schrödinger Operators","date":"2020-04-15","arxiv_id":"2004.06977","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploit-where-optimizer-explores-via","title":"Detached Error Feedback for Distributed SGD with Random Sparsification","date":"2020-04-11","arxiv_id":"2004.05298","n_code_links":0,"syntology":null},{"paper":"/paper/fda-fourier-domain-adaptation-for-semantic","slug":"fda-fourier-domain-adaptation-for-semantic","title":"FDA: Fourier Domain Adaptation for Semantic Segmentation","date":"2020-04-11","arxiv_id":"2004.05498","n_code_links":3,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["YanchaoYang/FDA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"continuous-and-discrete-time-analysis-of","title":"Convergence rates and approximation results for SGD and its continuous-time counterpart","date":"2020-04-08","arxiv_id":"2004.04193","n_code_links":0,"syntology":null},{"paper":null,"slug":"stopping-criteria-for-and-strong-convergence","title":"Stopping Criteria for, and Strong Convergence of, Stochastic Gradient Descent on Bottou-Curtis-Nocedal Functions","date":"2020-04-01","arxiv_id":"2004.00475","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-simple-class-decision-balancing-for","title":"SS-IL: Separated Softmax for Incremental Learning","date":"2020-03-31","arxiv_id":"2003.13947","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-proximal-gradient-algorithm-with","title":"Stochastic Proximal Gradient Algorithm with Minibatches. Application to Large Scale Learning Models","date":"2020-03-30","arxiv_id":"2003.13332","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-hybrid-order-distributed-sgd-method-for-non","title":"A Hybrid-Order Distributed SGD Method for Non-Convex Optimization to Balance Communication Overhead, Computational Complexity, and Convergence Rate","date":"2020-03-27","arxiv_id":"2003.12423","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-optimization-dynamics-of-wide","slug":"on-the-optimization-dynamics-of-wide","title":"On Infinite-Width Hypernetworks","date":"2020-03-27","arxiv_id":"2003.12193","n_code_links":1,"syntology":null},{"paper":null,"slug":"convergence-of-recursive-stochastic","title":"Convergence of Recursive Stochastic Algorithms using Wasserstein Divergence","date":"2020-03-25","arxiv_id":"2003.11403","n_code_links":0,"syntology":null},{"paper":"/paper/data-parallelism-in-training-sparse-neural","slug":"data-parallelism-in-training-sparse-neural","title":"Understanding the Effects of Data Parallelism and Sparsity on Neural Network Training","date":"2020-03-25","arxiv_id":"2003.11316","n_code_links":0,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/pipelined-backpropagation-at-scale-training","slug":"pipelined-backpropagation-at-scale-training","title":"Pipelined Backpropagation at Scale: Training Large Models without Batches","date":"2020-03-25","arxiv_id":"2003.11666","n_code_links":0,"syntology":null},{"paper":null,"slug":"fedsel-federated-sgd-under-local-differential","title":"FedSel: Federated SGD under Local Differential Privacy with Top-k Dimension Selection","date":"2020-03-24","arxiv_id":"2003.10637","n_code_links":0,"syntology":null},{"paper":null,"slug":"finite-time-analysis-of-stochastic-gradient","title":"Finite-Time Analysis of Stochastic Gradient Descent under Markov Randomness","date":"2020-03-24","arxiv_id":"2003.10973","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-classification-for-the-performance-of","title":"Online stochastic gradient descent on non-convex losses from high-dimensional inference","date":"2020-03-23","arxiv_id":"2003.10409","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-unified-theory-of-decentralized-sgd-with","title":"A Unified Theory of Decentralized SGD with Changing Topology and Local Updates","date":"2020-03-23","arxiv_id":"2003.10422","n_code_links":0,"syntology":null},{"paper":null,"slug":"slow-and-stale-gradients-can-win-the-race-1","title":"Slow and Stale Gradients Can Win the Race","date":"2020-03-23","arxiv_id":"2003.10579","n_code_links":0,"syntology":null},{"paper":null,"slug":"necpd-an-online-tensor-decomposition-with","title":"NeCPD: An Online Tensor Decomposition with Optimal Stochastic Gradient Descent","date":"2020-03-18","arxiv_id":"2003.08844","n_code_links":0,"syntology":null},{"paper":null,"slug":"explaining-memorization-and-generalization-a","title":"Weak and Strong Gradient Directions: Explaining Memorization, Generalization, and Hardness of Examples at Scale","date":"2020-03-16","arxiv_id":"2003.07422","n_code_links":0,"syntology":null},{"paper":"/paper/investigating-generalization-in-neural","slug":"investigating-generalization-in-neural","title":"Investigating Generalization in Neural Networks under Optimally Evolved Training Perturbations","date":"2020-03-14","arxiv_id":"2003.06646","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-implicit-bias-explain-generalization","title":"Can Implicit Bias Explain Generalization? Stochastic Convex Optimization as a Case Study","date":"2020-03-13","arxiv_id":"2003.06152","n_code_links":0,"syntology":null},{"paper":null,"slug":"machine-learning-on-volatile-instances","title":"Machine Learning on Volatile Instances","date":"2020-03-12","arxiv_id":"2003.05649","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-mean-field-analysis-of-deep-resnet-and","title":"A Mean-field Analysis of Deep ResNet and Beyond: Towards Provable Optimization Via Overparameterization From Depth","date":"2020-03-11","arxiv_id":"2003.05508","n_code_links":0,"syntology":null},{"paper":null,"slug":"shadowsync-performing-synchronization-in-the","title":"ShadowSync: Performing Synchronization in the Background for Highly Scalable Distributed Training","date":"2020-03-07","arxiv_id":"2003.03477","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-adam-and-adagrad","title":"A Simple Convergence Proof of Adam and Adagrad","date":"2020-03-05","arxiv_id":"2003.02395","n_code_links":0,"syntology":null},{"paper":"/paper/neural-kernels-without-tangents","slug":"neural-kernels-without-tangents","title":"Neural Kernels Without Tangents","date":"2020-03-04","arxiv_id":"2003.02237","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"basgd-buffered-asynchronous-sgd-for-byzantine","title":"Buffered Asynchronous SGD for Byzantine Learning","date":"2020-03-02","arxiv_id":"2003.00937","n_code_links":0,"syntology":null},{"paper":null,"slug":"iterate-averaging-helps-an-alternative","title":"Iterative Averaging in the Quest for Best Test Error","date":"2020-03-02","arxiv_id":"2003.01247","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-global-convergence-of-training-deep-1","title":"On the Global Convergence of Training Deep Linear ResNets","date":"2020-03-02","arxiv_id":"2003.01094","n_code_links":0,"syntology":null},{"paper":null,"slug":"toward-a-theory-of-optimization-for-over","title":"Loss landscapes and optimization in over-parameterized non-linear systems and neural networks","date":"2020-02-29","arxiv_id":"2003.00307","n_code_links":0,"syntology":null},{"paper":"/paper/distributed-momentum-for-byzantine-resilient","slug":"distributed-momentum-for-byzantine-resilient","title":"Distributed Momentum for Byzantine-resilient Learning","date":"2020-02-28","arxiv_id":"2003.00010","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":3,"phrase":"0 ran · 3 unverified","official":{"repos":["LPD-EPFL/ByzantineMomentum"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"paper":null,"slug":"do-optimization-methods-in-deep-learning","title":"Do optimization methods in deep learning applications matter?","date":"2020-02-28","arxiv_id":"2002.12642","n_code_links":0,"syntology":null},{"paper":"/paper/fast-and-three-rious-speeding-up-weak","slug":"fast-and-three-rious-speeding-up-weak","title":"Fast and Three-rious: Speeding Up Weak Supervision with Triplet Methods","date":"2020-02-27","arxiv_id":"2002.11955","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-biased-compression-for-distributed","title":"On Biased Compression for Distributed Learning","date":"2020-02-27","arxiv_id":"2002.12410","n_code_links":0,"syntology":null},{"paper":"/paper/lasg-lazily-aggregated-stochastic-gradients","slug":"lasg-lazily-aggregated-stochastic-gradients","title":"LASG: Lazily Aggregated Stochastic Gradients for Communication-Efficient Distributed Learning","date":"2020-02-26","arxiv_id":"2002.11360","n_code_links":1,"syntology":null},{"paper":null,"slug":"moniqua-modulo-quantized-communication-in-1","title":"Moniqua: Modulo Quantized Communication in Decentralized SGD","date":"2020-02-26","arxiv_id":"2002.11787","n_code_links":0,"syntology":null},{"paper":null,"slug":"non-asymptotic-bounds-for-zeroth-order","title":"Non-asymptotic bounds for stochastic optimization with biased noisy gradient oracles","date":"2020-02-26","arxiv_id":"2002.11440","n_code_links":0,"syntology":null},{"paper":null,"slug":"stagewise-enlargement-of-batch-size-for-sgd","title":"Stagewise Enlargement of Batch Size for SGD-based Learning","date":"2020-02-26","arxiv_id":"2002.11601","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-distributed-stochastic-gradient","title":"Adaptive Distributed Stochastic Gradient Descent for Minimizing Delay in the Presence of Stragglers","date":"2020-02-25","arxiv_id":"2002.11005","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-sign-sgd-for-federated-learning","title":"Stochastic-Sign SGD for Federated Learning with Theoretical Guarantees","date":"2020-02-25","arxiv_id":"2002.10940","n_code_links":0,"syntology":null},{"paper":null,"slug":"closing-the-convergence-gap-of-sgd-without","title":"Closing the convergence gap of SGD without replacement","date":"2020-02-24","arxiv_id":"2002.10400","n_code_links":0,"syntology":null},{"paper":"/paper/on-pruning-adversarially-robust-neural","slug":"on-pruning-adversarially-robust-neural","title":"HYDRA: Pruning Adversarially Robust Neural Networks","date":"2020-02-24","arxiv_id":"2002.10509","n_code_links":4,"syntology":{"ran":4,"of":19,"n_ran_checked":4,"n_instrument":0,"unverified":15,"pointer_only":17,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 15 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":{"repos":["inspire-group/compactness-robustness"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/scheduled-restart-momentum-for-accelerated","slug":"scheduled-restart-momentum-for-accelerated","title":"Scheduled Restart Momentum for Accelerated Stochastic Gradient Descent","date":"2020-02-24","arxiv_id":"2002.10583","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["minhtannguyen/SRSGD"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/stochastic-polyak-step-size-for-sgd-an","slug":"stochastic-polyak-step-size-for-sgd-an","title":"Stochastic Polyak Step-size for SGD: An Adaptive Learning Rate for Fast Convergence","date":"2020-02-24","arxiv_id":"2002.10542","n_code_links":1,"syntology":null},{"paper":null,"slug":"improve-sgd-training-via-aligning-min-batches","title":"Improve SGD Training via Aligning Mini-batches","date":"2020-02-23","arxiv_id":"2002.09917","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-continually-learn","slug":"learning-to-continually-learn","title":"Learning to Continually Learn","date":"2020-02-21","arxiv_id":"2002.09571","n_code_links":5,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["uvm-neurobotics-lab/ANML"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/overlap-local-sgd-an-algorithmic-approach-to","slug":"overlap-local-sgd-an-algorithmic-approach-to","title":"Overlap Local-SGD: An Algorithmic Approach to Hide Communication Delays in Distributed SGD","date":"2020-02-21","arxiv_id":"2002.09539","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-break-even-point-on-optimization","title":"The Break-Even Point on Optimization Trajectories of Deep Neural Networks","date":"2020-02-21","arxiv_id":"2002.09572","n_code_links":0,"syntology":null},{"paper":null,"slug":"bounding-the-expected-run-time-of-nonconvex","title":"Bounding the expected run-time of nonconvex optimization with early stopping","date":"2020-02-20","arxiv_id":"2002.08856","n_code_links":0,"syntology":null},{"paper":null,"slug":"embedding-graph-auto-encoder-with-joint","title":"Embedding Graph Auto-Encoder for Graph Clustering","date":"2020-02-20","arxiv_id":"2002.08643","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-runge-kutta-methods-and-adaptive","title":"Stochastic Runge-Kutta methods and adaptive SGD-G2 stochastic gradient descent","date":"2020-02-20","arxiv_id":"2002.09304","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributed-optimization-over-block-cyclic","title":"Distributed Optimization over Block-Cyclic Data","date":"2020-02-18","arxiv_id":"2002.07454","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-local-sgd-better-than-minibatch-sgd","title":"Is Local SGD Better than Minibatch SGD?","date":"2020-02-18","arxiv_id":"2002.07839","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-convergence-for-langevin-diffusion-with","title":"Fast Convergence for Langevin Diffusion with Manifold Structure","date":"2020-02-13","arxiv_id":"2002.05576","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-halfspaces-with-massart-noise-under","title":"Learning Halfspaces with Massart Noise Under Structured Distributions","date":"2020-02-13","arxiv_id":"2002.05632","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-diffusion-theory-for-deep-learning-dynamics","title":"A Diffusion Theory For Deep Learning Dynamics: Stochastic Gradient Descent Exponentially Favors Flat Minima","date":"2020-02-10","arxiv_id":"2002.03495","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-fully-online-approach-for-covariance","title":"Online Covariance Matrix Estimation in Stochastic Gradient Descent","date":"2020-02-10","arxiv_id":"2002.03979","n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-learning-of-a-mixture-of-global-and","title":"Federated Learning of a Mixture of Global and Local Models","date":"2020-02-10","arxiv_id":"2002.05516","n_code_links":0,"syntology":null},{"paper":null,"slug":"semi-implicit-back-propagation-1","title":"Semi-Implicit Back Propagation","date":"2020-02-10","arxiv_id":"2002.03516","n_code_links":0,"syntology":null},{"paper":null,"slug":"better-theory-for-sgd-in-the-nonconvex-world","title":"Better Theory for SGD in the Nonconvex World","date":"2020-02-09","arxiv_id":"2002.03329","n_code_links":0,"syntology":null},{"paper":null,"slug":"momentum-improves-normalized-sgd","title":"Momentum Improves Normalized SGD","date":"2020-02-09","arxiv_id":"2002.03305","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-distance-between-two-neural-networks","slug":"on-the-distance-between-two-neural-networks","title":"On the distance between two neural networks and the stability of learning","date":"2020-02-09","arxiv_id":"2002.03432","n_code_links":2,"syntology":null},{"paper":"/paper/how-good-is-the-bayes-posterior-in-deep","slug":"how-good-is-the-bayes-posterior-in-deep","title":"How Good is the Bayes Posterior in Deep Neural Networks Really?","date":"2020-02-06","arxiv_id":"2002.02405","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-mean-field-theory-of-lazy-training-in-two","title":"Function approximation by neural nets in the mean-field regime: Entropic regularization and controlled McKean-Vlasov dynamics","date":"2020-02-05","arxiv_id":"2002.01987","n_code_links":0,"syntology":null},{"paper":null,"slug":"goal-oriented-multi-task-bert-based-dialogue","title":"Goal-Oriented Multi-Task BERT-Based Dialogue State Tracker","date":"2020-02-05","arxiv_id":"2002.02450","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-riemannian-optimization-on-the-1","title":"Efficient Riemannian Optimization on the Stiefel Manifold via the Cayley Transform","date":"2020-02-04","arxiv_id":"2002.01113","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-efficiency-in-large-scale","title":"Improving Efficiency in Large-Scale Decentralized Distributed Training","date":"2020-02-04","arxiv_id":"2002.01119","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-stochastic-gradient-3","title":"On the Convergence of Stochastic Gradient Descent with Low-Rank Projections for Convex Low-Rank Matrix Problems","date":"2020-01-31","arxiv_id":"2001.11668","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-does-bn-increase-collapsed-neural-network","title":"How Does BN Increase Collapsed Neural Network Filters?","date":"2020-01-30","arxiv_id":"2001.11216","n_code_links":0,"syntology":null},{"paper":null,"slug":"variance-reduction-with-sparse-gradients-1","title":"Variance Reduction with Sparse Gradients","date":"2020-01-27","arxiv_id":"2001.09623","n_code_links":0,"syntology":null},{"paper":"/paper/stochastic-optimization-of-plain","slug":"stochastic-optimization-of-plain","title":"Stochastic Optimization of Plain Convolutional Neural Networks with Simple methods","date":"2020-01-24","arxiv_id":"2001.08856","n_code_links":1,"syntology":null},{"paper":null,"slug":"intermittent-pulling-with-local-compensation","title":"Intermittent Pulling with Local Compensation for Communication-Efficient Federated Learning","date":"2020-01-22","arxiv_id":"2001.08277","n_code_links":0,"syntology":null},{"paper":"/paper/on-last-layer-algorithms-for-classification","slug":"on-last-layer-algorithms-for-classification","title":"On Last-Layer Algorithms for Classification: Decoupling Representation from Uncertainty Estimation","date":"2020-01-22","arxiv_id":"2001.08049","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nbrosse/uncertainties"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"stochastic-item-descent-method-for-large","title":"Stochastic Item Descent Method for Large Scale Equal Circle Packing Problem","date":"2020-01-22","arxiv_id":"2001.08540","n_code_links":0,"syntology":null},{"paper":"/paper/harmonic-convolutional-networks-based-on","slug":"harmonic-convolutional-networks-based-on","title":"Harmonic Convolutional Networks based on Discrete Cosine Transform","date":"2020-01-18","arxiv_id":"2001.06570","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":10,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["matej-ulicny/harmonic-networks"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"elastic-consistency-a-general-consistency","title":"Elastic Consistency: A General Consistency Model for Distributed Stochastic Gradient Descent","date":"2020-01-16","arxiv_id":"2001.05918","n_code_links":0,"syntology":null},{"paper":null,"slug":"backward-feature-correction-how-deep-learning","title":"Backward Feature Correction: How Deep Learning Performs Deep (Hierarchical) Learning","date":"2020-01-13","arxiv_id":"2001.04413","n_code_links":0,"syntology":null},{"paper":"/paper/choosing-the-sample-with-lowest-loss-makes","slug":"choosing-the-sample-with-lowest-loss-makes","title":"Choosing the Sample with Lowest Loss makes SGD Robust","date":"2020-01-10","arxiv_id":"2001.03316","n_code_links":1,"syntology":null},{"paper":"/paper/sgd-with-hardness-weighted-sampling-for-1","slug":"sgd-with-hardness-weighted-sampling-for-1","title":"Distributionally Robust Deep Learning using Hardness Weighted Sampling","date":"2020-01-08","arxiv_id":"2001.02658","n_code_links":1,"syntology":null},{"paper":null,"slug":"poly-time-universality-and-limitations-of","title":"Poly-time universality and limitations of deep learning","date":"2020-01-07","arxiv_id":"2001.02992","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-neural-networks-find-generalizable","title":"How neural networks find generalizable solutions: Self-tuned annealing in deep learning","date":"2020-01-06","arxiv_id":"2001.01678","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-understanding-the-true-loss-surface","title":"Towards understanding the true loss surface of deep neural networks using random matrix theory and iterative spectral methods","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"training-deep-networks-with-stochastic","title":"Training Deep Networks with Stochastic Gradient Normalized by Layerwise Adaptive Second Moments","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-dynamic-sampling-adaptive-sgd-method-for","title":"A Dynamic Sampling Adaptive-SGD Method for Machine Learning","date":"2019-12-31","arxiv_id":"1912.13357","n_code_links":0,"syntology":null},{"paper":"/paper/variance-reduced-local-sgd-with-lower-1","slug":"variance-reduced-local-sgd-with-lower-1","title":"Variance Reduced Local SGD with Lower Communication Complexity","date":"2019-12-30","arxiv_id":"1912.12844","n_code_links":1,"syntology":null},{"paper":null,"slug":"federated-variance-reduced-stochastic","title":"Federated Variance-Reduced Stochastic Gradient Descent with Robustness to Byzantine Attacks","date":"2019-12-29","arxiv_id":"1912.12716","n_code_links":0,"syntology":null},{"paper":"/paper/cprop-adaptive-learning-rate-scaling-from","slug":"cprop-adaptive-learning-rate-scaling-from","title":"CProp: Adaptive Learning Rate Scaling from Past Gradient Conformity","date":"2019-12-24","arxiv_id":"1912.11493","n_code_links":1,"syntology":null},{"paper":null,"slug":"landscape-connectivity-and-dropout-stability","title":"Landscape Connectivity and Dropout Stability of SGD Solutions for Over-parameterized Neural Networks","date":"2019-12-20","arxiv_id":"1912.10095","n_code_links":0,"syntology":null},{"paper":null,"slug":"second-order-information-in-first-order","title":"Second-order Information in First-order Optimization Methods","date":"2019-12-20","arxiv_id":"1912.09926","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimization-for-deep-learning-theory-and","title":"Optimization for deep learning: theory and algorithms","date":"2019-12-19","arxiv_id":"1912.08957","n_code_links":0,"syntology":null},{"paper":"/paper/gradient-based-training-of-gaussian-mixture-1","slug":"gradient-based-training-of-gaussian-mixture-1","title":"Gradient-based training of Gaussian Mixture Models for High-Dimensional Streaming Data","date":"2019-12-18","arxiv_id":"1912.09379","n_code_links":1,"syntology":null},{"paper":"/paper/generative-teaching-networks-accelerating-1","slug":"generative-teaching-networks-accelerating-1","title":"Generative Teaching Networks: Accelerating Neural Architecture Search by Learning to Generate Synthetic Training Data","date":"2019-12-17","arxiv_id":"1912.07768","n_code_links":3,"syntology":null},{"paper":null,"slug":"parallel-restarted-spider-communication","title":"Parallel Restarted SPIDER -- Communication Efficient Distributed Nonconvex Optimization with Optimal Computation Complexity","date":"2019-12-12","arxiv_id":"1912.06036","n_code_links":0,"syntology":null},{"paper":"/paper/linear-mode-connectivity-and-the-lottery","slug":"linear-mode-connectivity-and-the-lottery","title":"Linear Mode Connectivity and the Lottery Ticket Hypothesis","date":"2019-12-11","arxiv_id":"1912.05671","n_code_links":2,"syntology":null},{"paper":null,"slug":"siamman-siamese-motion-aware-network-for","title":"SiamMan: Siamese Motion-aware Network for Visual Tracking","date":"2019-12-11","arxiv_id":"1912.05515","n_code_links":0,"syntology":null},{"paper":null,"slug":"why-adam-beats-sgd-for-attention-models-1","title":"Why are Adaptive Methods Good for Attention Models?","date":"2019-12-06","arxiv_id":"1912.03194","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-intrinsic-privacy-of-stochastic","slug":"on-the-intrinsic-privacy-of-stochastic","title":"An Empirical Study on the Intrinsic Privacy of SGD","date":"2019-12-05","arxiv_id":"1912.02919","n_code_links":1,"syntology":{"ran":11,"of":16,"n_ran_checked":11,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["microsoft/intrinsic-private-sgd"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/domain-independent-dominance-of-adaptive-1","slug":"domain-independent-dominance-of-adaptive-1","title":"Domain-independent Dominance of Adaptive Methods","date":"2019-12-04","arxiv_id":"1912.01823","n_code_links":1,"syntology":null},{"paper":null,"slug":"stochastic-variational-inference-via-upper","title":"Stochastic Variational Inference via Upper Bound","date":"2019-12-02","arxiv_id":"1912.00650","n_code_links":0,"syntology":null}],"record_sha256":"9ff25dbfdc1992ae9a624b083a83167625fa0062a3be57f556c3041da6990654","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}