{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sgd/papers/9","list_of":"/method/sgd","method":"SGD","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":9,"pages_in_order":21,"rows_per_page":100,"rows":[801,900],"of":2021,"counts":{"archive_papers_tagged":2021,"with_a_code_link":591,"where_syntology_ran_a_sample":192,"not_listed_spam_title":0,"listed":2021,"listed_where_code_ran":192,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sgd","prev":"/method/sgd/papers/8","next":"/method/sgd/papers/10","papers":[{"paper":null,"slug":"locally-asynchronous-stochastic-gradient","title":"Locally Asynchronous Stochastic Gradient Descent for Decentralised Deep Learning","date":"2022-03-24","arxiv_id":"2203.13085","n_code_links":0,"syntology":null},{"paper":"/paper/on-exploiting-layerwise-gradient-statistics","slug":"on-exploiting-layerwise-gradient-statistics","title":"A DNN Optimizer that Improves over AdaBelief by Suppression of the Adaptive Stepsize Range","date":"2022-03-24","arxiv_id":"2203.13273","n_code_links":1,"syntology":null},{"paper":"/paper/an-adaptive-gradient-method-with-energy-and","slug":"an-adaptive-gradient-method-with-energy-and","title":"An Adaptive Gradient Method with Energy and Momentum","date":"2022-03-23","arxiv_id":"2203.12191","n_code_links":1,"syntology":null},{"paper":"/paper/thingtalk-an-extensible-executable","slug":"thingtalk-an-extensible-executable","title":"ThingTalk: An Extensible, Executable Representation Language for Task-Oriented Dialogues","date":"2022-03-23","arxiv_id":"2203.12751","n_code_links":1,"syntology":null},{"paper":"/paper/practical-tradeoffs-between-memory-compute","slug":"practical-tradeoffs-between-memory-compute","title":"Practical tradeoffs between memory, compute, and performance in learned optimizers","date":"2022-03-22","arxiv_id":"2203.11860","n_code_links":1,"syntology":null},{"paper":null,"slug":"provable-constrained-stochastic-convex","title":"Provable Constrained Stochastic Convex Optimization with XOR-Projected Gradient Descent","date":"2022-03-22","arxiv_id":"2203.11829","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-local-convergence-theory-for-the-stochastic","title":"A Local Convergence Theory for the Stochastic Gradient Descent Method in Non-Convex Optimization With Non-isolated Local Minima","date":"2022-03-21","arxiv_id":"2203.10973","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-perspective-on-probabilistic-image","title":"A new perspective on probabilistic image modeling","date":"2022-03-21","arxiv_id":"2203.11034","n_code_links":0,"syntology":null},{"paper":null,"slug":"imagenet-challenging-classification-with-the","title":"ImageNet Challenging Classification with the Raspberry Pi: An Incremental Local Stochastic Gradient Descent Algorithm","date":"2022-03-21","arxiv_id":"2203.11853","n_code_links":0,"syntology":null},{"paper":null,"slug":"fake-news-detection-using-majority-voting","title":"Fake News Detection Using Majority Voting Technique","date":"2022-03-18","arxiv_id":"2203.09936","n_code_links":0,"syntology":null},{"paper":null,"slug":"ss-sam-stochastic-scheduled-sharpness-aware","title":"Randomized Sharpness-Aware Training for Boosting Computational Efficiency in Deep Learning","date":"2022-03-18","arxiv_id":"2203.09962","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-role-of-local-steps-in-local-sgd","title":"The Role of Local Steps in Local SGD","date":"2022-03-14","arxiv_id":"2203.06798","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-the-wild-decentralizing-hogwild-style","slug":"scaling-the-wild-decentralizing-hogwild-style","title":"Scaling the Wild: Decentralizing Hogwild!-style Shared-memory SGD","date":"2022-03-13","arxiv_id":"2203.06638","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-adversarial-training-with-second","slug":"enhancing-adversarial-training-with-second","title":"Enhancing Adversarial Training with Second-Order Statistics of Weights","date":"2022-03-11","arxiv_id":"2203.06020","n_code_links":1,"syntology":null},{"paper":null,"slug":"differentially-private-learning-needs-hidden","title":"Differentially Private Learning Needs Hidden State (Or Much Faster Convergence)","date":"2022-03-10","arxiv_id":"2203.05363","n_code_links":0,"syntology":null},{"paper":null,"slug":"risk-bounds-of-multi-pass-sgd-for-least","title":"Risk Bounds of Multi-Pass SGD for Least Squares in the Interpolation Regime","date":"2022-03-07","arxiv_id":"2203.03159","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-did-you-say-task-oriented-dialog","title":"What Did You Say? Task-Oriented Dialog Datasets Are Not Conversational!?","date":"2022-03-07","arxiv_id":"2203.03431","n_code_links":0,"syntology":null},{"paper":"/paper/towards-efficient-and-scalable-sharpness","slug":"towards-efficient-and-scalable-sharpness","title":"Towards Efficient and Scalable Sharpness-Aware Minimization","date":"2022-03-05","arxiv_id":"2203.02714","n_code_links":4,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":null}},{"paper":null,"slug":"distributed-methods-with-absolute-compression","title":"Distributed Methods with Absolute Compression and Error Compensation","date":"2022-03-04","arxiv_id":"2203.02383","n_code_links":0,"syntology":null},{"paper":null,"slug":"benign-underfitting-of-stochastic-gradient","title":"Benign Underfitting of Stochastic Gradient Descent","date":"2022-02-27","arxiv_id":"2202.13361","n_code_links":0,"syntology":null},{"paper":null,"slug":"explicit-regularization-via-regularizer","title":"Explicit Regularization via Regularizer Mirror Descent","date":"2022-02-22","arxiv_id":"2202.10788","n_code_links":0,"syntology":null},{"paper":null,"slug":"personalized-federated-learning-with-exact","title":"Personalized Federated Learning with Exact Stochastic Gradient Descent","date":"2022-02-20","arxiv_id":"2202.09848","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-note-on-the-implicit-bias-towards-minimal","title":"On the Implicit Bias Towards Minimal Depth of Deep Neural Networks","date":"2022-02-18","arxiv_id":"2202.09028","n_code_links":0,"syntology":null},{"paper":null,"slug":"tackling-benign-nonconvexity-with-smoothing","title":"Tackling benign nonconvexity with smoothing and stochastic gradients","date":"2022-02-18","arxiv_id":"2202.09052","n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-stochastic-gradient-descent-begets","title":"Federated Stochastic Gradient Descent Begets Self-Induced Momentum","date":"2022-02-17","arxiv_id":"2202.08402","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-merged-staircase-property-a-necessary-and","title":"The merged-staircase property: a necessary and nearly sufficient condition for SGD learning of sparse functions on two-layer neural networks","date":"2022-02-17","arxiv_id":"2202.08658","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-distributed-machine-learning-via","title":"Cost-Efficient Distributed Learning via Combinatorial Multi-Armed Bandits","date":"2022-02-16","arxiv_id":"2202.08302","n_code_links":0,"syntology":null},{"paper":null,"slug":"black-box-generalization","title":"Black-Box Generalization: Stability of Zeroth-Order Learning","date":"2022-02-14","arxiv_id":"2202.06880","n_code_links":0,"syntology":null},{"paper":"/paper/orthogonalising-gradients-to-speed-up-neural","slug":"orthogonalising-gradients-to-speed-up-neural","title":"Orthogonalising gradients to speed up neural network optimisation","date":"2022-02-14","arxiv_id":"2202.07052","n_code_links":1,"syntology":null},{"paper":null,"slug":"escaping-saddle-points-with-bias-variance","title":"Escaping Saddle Points with Bias-Variance Reduced Local Perturbed SGD for Communication Efficient Nonconvex Distributed Learning","date":"2022-02-12","arxiv_id":"2202.06083","n_code_links":0,"syntology":null},{"paper":"/paper/maximizing-communication-efficiency-for-large","slug":"maximizing-communication-efficiency-for-large","title":"Maximizing Communication Efficiency for Large-scale Training via 0/1 Adam","date":"2022-02-12","arxiv_id":"2202.06009","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-power-of-adaptivity-in-sgd-self-tuning","title":"The Power of Adaptivity in SGD: Self-Tuning Step Sizes with Unbounded Gradients and Affine Variance","date":"2022-02-11","arxiv_id":"2202.05791","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-linear-regression-for-general-feature","title":"Robust Linear Regression for General Feature Distribution","date":"2022-02-04","arxiv_id":"2202.02080","n_code_links":0,"syntology":null},{"paper":"/paper/signsgd-fault-tolerance-to-blind-and","slug":"signsgd-fault-tolerance-to-blind-and","title":"SignSGD: Fault-Tolerance to Blind and Byzantine Adversaries","date":"2022-02-04","arxiv_id":"2202.02085","n_code_links":3,"syntology":null},{"paper":null,"slug":"characterizing-finding-good-data-orderings","title":"Characterizing & Finding Good Data Orderings for Fast Convergence of Sequential Gradient Methods","date":"2022-02-03","arxiv_id":"2202.01838","n_code_links":0,"syntology":null},{"paper":"/paper/fast-convex-optimization-for-two-layer-relu","slug":"fast-convex-optimization-for-two-layer-relu","title":"Fast Convex Optimization for Two-Layer ReLU Networks: Equivalent Model Classes and Cone Decompositions","date":"2022-02-02","arxiv_id":"2202.01331","n_code_links":2,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["pilancilab/scnn"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"robust-training-of-neural-networks-using","title":"Robust Training of Neural Networks Using Scale Invariant Architectures","date":"2022-02-02","arxiv_id":"2202.00980","n_code_links":0,"syntology":null},{"paper":"/paper/phase-diagram-of-stochastic-gradient-descent","slug":"phase-diagram-of-stochastic-gradient-descent","title":"Phase diagram of Stochastic Gradient Descent in high-dimensional two-layer neural networks","date":"2022-02-01","arxiv_id":"2202.00293","n_code_links":2,"syntology":null},{"paper":null,"slug":"faster-convergence-of-local-sgd-for-over","title":"Faster Convergence of Local SGD for Over-Parameterized Models","date":"2022-01-30","arxiv_id":"2201.12719","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-simple-guard-for-learned-optimizers","title":"A Simple Guard for Learned Optimizers","date":"2022-01-28","arxiv_id":"2201.12426","n_code_links":0,"syntology":null},{"paper":"/paper/dropnas-grouped-operation-dropout-for","slug":"dropnas-grouped-operation-dropout-for","title":"DropNAS: Grouped Operation Dropout for Differentiable Architecture Search","date":"2022-01-27","arxiv_id":"2201.11679","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-msgd-and-adagrad-for-1","title":"On the Convergence of mSGD and AdaGrad for Stochastic Optimization","date":"2022-01-26","arxiv_id":"2201.11204","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-uniform-boundedness-properties-of-sgd-and","title":"On Uniform Boundedness Properties of SGD and its Momentum Variants","date":"2022-01-25","arxiv_id":"2201.10245","n_code_links":0,"syntology":null},{"paper":"/paper/description-driven-task-oriented-dialog-1","slug":"description-driven-task-oriented-dialog-1","title":"Description-Driven Task-Oriented Dialog Modeling","date":"2022-01-21","arxiv_id":"2201.08904","n_code_links":1,"syntology":null},{"paper":"/paper/low-pass-filtering-sgd-for-recovering-flat","slug":"low-pass-filtering-sgd-for-recovering-flat","title":"Low-Pass Filtering SGD for Recovering Flat Optima in the Deep Learning Optimization Landscape","date":"2022-01-20","arxiv_id":"2201.08025","n_code_links":1,"syntology":null},{"paper":"/paper/adaterm-adaptive-t-distribution-estimated","slug":"adaterm-adaptive-t-distribution-estimated","title":"AdaTerm: Adaptive T-Distribution Estimated Robust Moments for Noise-Robust Stochastic Gradient Optimization","date":"2022-01-18","arxiv_id":"2201.06714","n_code_links":1,"syntology":null},{"paper":null,"slug":"unsupervised-slot-schema-induction-for-task","title":"Unsupervised Slot Schema Induction for Task-oriented Dialog","date":"2022-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/on-generalization-bounds-for-deep-networks","slug":"on-generalization-bounds-for-deep-networks","title":"On generalization bounds for deep networks based on loss surface implicit regularization","date":"2022-01-12","arxiv_id":"2201.04545","n_code_links":1,"syntology":null},{"paper":null,"slug":"partial-model-averaging-in-federated-learning","title":"Partial Model Averaging in Federated Learning: Performance Guarantees and Benefits","date":"2022-01-11","arxiv_id":"2201.03789","n_code_links":0,"syntology":null},{"paper":null,"slug":"stability-based-generalization-bounds-for-1","title":"Stability Based Generalization Bounds for Exponential Family Langevin Dynamics","date":"2022-01-09","arxiv_id":"2201.03064","n_code_links":0,"syntology":null},{"paper":"/paper/the-dynamics-of-representation-learning-in","slug":"the-dynamics-of-representation-learning-in","title":"The dynamics of representation learning in shallow, non-linear autoencoders","date":"2022-01-06","arxiv_id":"2201.02115","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":{"repos":["mariaref/nonlinearshallowae"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/convergence-and-complexity-of-stochastic","slug":"convergence-and-complexity-of-stochastic","title":"Stochastic regularized majorization-minimization with weakly convex and multi-convex surrogates","date":"2022-01-05","arxiv_id":"2201.01652","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluation-of-thermal-imaging-on-embedded-gpu","title":"Evaluation of Thermal Imaging on Embedded GPU Platforms for Application in Vehicular Assistance Systems","date":"2022-01-05","arxiv_id":"2201.01661","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-mixed-integer-programming-approach-to","title":"A Mixed-Integer Programming Approach to Training Dense Neural Networks","date":"2022-01-03","arxiv_id":"2201.00723","n_code_links":0,"syntology":null},{"paper":"/paper/stochastic-weight-averaging-revisited","slug":"stochastic-weight-averaging-revisited","title":"Stochastic Weight Averaging Revisited","date":"2022-01-03","arxiv_id":"2201.00519","n_code_links":1,"syntology":null},{"paper":null,"slug":"accelerating-neural-network-optimization","title":"Accelerating Neural Network Optimization Through an Automated Control Theory Lens","date":"2022-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"distributed-and-stochastic-optimization","title":"Distributed and Stochastic Optimization Methods with Gradient Compression and Local Steps","date":"2021-12-20","arxiv_id":"2112.10645","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-effective-noise-of-stochastic-gradient","title":"The effective noise of Stochastic Gradient Descent","date":"2021-12-20","arxiv_id":"2112.10852","n_code_links":0,"syntology":null},{"paper":"/paper/adaptively-customizing-activation-functions","slug":"adaptively-customizing-activation-functions","title":"Adaptively Customizing Activation Functions for Various Layers","date":"2021-12-17","arxiv_id":"2112.09442","n_code_links":1,"syntology":null},{"paper":null,"slug":"non-asymptotic-bounds-for-optimization-via","title":"Non-Asymptotic Analysis of Online Multiplicative Stochastic Gradient Descent","date":"2021-12-14","arxiv_id":"2112.07110","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-proof-for-stochastic-gradient","title":"Convergence proof for stochastic gradient descent in the training of deep neural networks with ReLU activation for constant target functions","date":"2021-12-13","arxiv_id":"2112.07369","n_code_links":0,"syntology":null},{"paper":null,"slug":"determinantal-point-processes-based-on-1","title":"Determinantal point processes based on orthogonal polynomials for sampling minibatches in SGD","date":"2021-12-11","arxiv_id":"2112.06007","n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-two-stage-learning-with-sign-based","title":"Federated Two-stage Learning with Sign-based Voting","date":"2021-12-10","arxiv_id":"2112.05687","n_code_links":0,"syntology":null},{"paper":null,"slug":"dr3-value-based-deep-reinforcement-learning-1","title":"DR3: Value-Based Deep Reinforcement Learning Requires Explicit Regularization","date":"2021-12-09","arxiv_id":"2112.04716","n_code_links":0,"syntology":null},{"paper":null,"slug":"fastsgd-a-fast-compressed-sgd-framework-for","title":"FastSGD: A Fast Compressed SGD Framework for Distributed Machine Learning","date":"2021-12-08","arxiv_id":"2112.04291","n_code_links":0,"syntology":null},{"paper":"/paper/training-structured-neural-networks-through-1","slug":"training-structured-neural-networks-through-1","title":"Training Structured Neural Networks Through Manifold Identification and Variance Reduction","date":"2021-12-05","arxiv_id":"2112.02612","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zihsyuan1214/rmda"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"loss-landscape-dependent-self-adjusting","title":"Loss Landscape Dependent Self-Adjusting Learning Rates in Decentralized Stochastic Gradient Descent","date":"2021-12-02","arxiv_id":"2112.01433","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-large-batch-training-and-sharp-minima-a","title":"On Large Batch Training and Sharp Minima: A Fokker-Planck Perspective","date":"2021-12-02","arxiv_id":"2112.00987","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-proximal-gradient-methods-for","title":"Adaptive Proximal Gradient Methods for Structured Neural Networks","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"an-improved-analysis-and-rates-for-variance","title":"An Improved Analysis and Rates for Variance Reduction under Without-replacement Sampling Orders","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"closing-the-gap-tighter-analysis-of","title":"Closing the Gap: Tighter Analysis of Alternating Stochastic Gradient Methods for Bilevel Problems","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"generalization-guarantee-of-sgd-for-pairwise","title":"Generalization Guarantee of SGD for Pairwise Learning","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"simple-stochastic-and-online-gradient-descent","title":"Simple Stochastic and Online Gradient Descent Algorithms for Pairwise Learning","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-implicit-bias-of-minima-stability-a-view","title":"The Implicit Bias of Minima Stability: A View from Function Space","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/towards-understanding-why-lookahead","slug":"towards-understanding-why-lookahead","title":"Towards Understanding Why Lookahead Generalizes Better Than SGD and Beyond","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"autodrop-training-deep-learning-models-with-1","title":"AutoDrop: Training Deep Learning Models with Automatic Learning Rate Drop","date":"2021-11-30","arxiv_id":"2111.15317","n_code_links":0,"syntology":null},{"paper":null,"slug":"randomized-stochastic-gradient-descent-ascent","title":"Randomized Stochastic Gradient Descent Ascent","date":"2021-11-25","arxiv_id":"2111.13162","n_code_links":0,"syntology":null},{"paper":"/paper/simple-stochastic-and-online-gradient","slug":"simple-stochastic-and-online-gradient","title":"Simple Stochastic and Online Gradient DescentAlgorithms for Pairwise Learning","date":"2021-11-23","arxiv_id":"2111.12050","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhenhuan-yang/simple-pairwise"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gaussian-process-inference-using-mini-batch","title":"Gaussian Process Inference Using Mini-batch Stochastic Gradient Descent: Convergence Guarantees and Empirical Benefits","date":"2021-11-19","arxiv_id":"2111.10461","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-asynchronous-distributed-training","title":"An asynchronous distributed training algorithm based on Gossip","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"description-driven-task-oriented-dialog","title":"Description-Driven Task-Oriented Dialog Modeling","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-schema-graph-fusion-network-for-multi","title":"Dynamic Schema Graph Fusion Network for Multi-Domain Dialogue State Tracking","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/image-specific-convolutional-kernel","slug":"image-specific-convolutional-kernel","title":"Image-specific Convolutional Kernel Modulation for Single Image Super-resolution","date":"2021-11-16","arxiv_id":"2111.08362","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-compositional-generalization-with-1","title":"Improving Compositional Generalization with Self-Training for Data-to-Text Generation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"online-meta-adaptation-for-variable-rate","title":"Online Meta Adaptation for Variable-Rate Learned Image Compression","date":"2021-11-16","arxiv_id":"2111.08256","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-schema-induction-for-task","title":"Self-supervised Schema Induction for Task-oriented Dialog","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"qk-iteration-a-self-supervised-representation","title":"QK Iteration: A Self-Supervised Representation Learning Algorithm for Image Similarity","date":"2021-11-15","arxiv_id":"2111.07954","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-gradient-line-bayesian","title":"Stochastic Gradient Line Bayesian Optimization for Efficient Noise-Robust Optimization of Parameterized Quantum Circuits","date":"2021-11-15","arxiv_id":"2111.07952","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-three-stages-of-learning-dynamics-in-high-1","title":"The Three Stages of Learning Dynamics in High-Dimensional Kernel Methods","date":"2021-11-13","arxiv_id":"2111.07167","n_code_links":0,"syntology":null},{"paper":null,"slug":"stationary-behavior-of-constant-stepsize-sgd","title":"Stationary Behavior of Constant Stepsize SGD Type Algorithms: An Asymptotic Characterization","date":"2021-11-11","arxiv_id":"2111.06328","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgd-through-the-lens-of-kolmogorov-complexity","title":"SGD Through the Lens of Kolmogorov Complexity","date":"2021-11-10","arxiv_id":"2111.05478","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-rectify-for-robust-learning-with","slug":"learning-to-rectify-for-robust-learning-with","title":"Learning to Rectify for Robust Learning with Noisy Labels","date":"2021-11-08","arxiv_id":"2111.04239","n_code_links":1,"syntology":null},{"paper":"/paper/quasi-potential-theory-for-escape-problem-1","slug":"quasi-potential-theory-for-escape-problem-1","title":"Exponential escape efficiency of SGD from sharp minima in non-stationary regime","date":"2021-11-07","arxiv_id":"2111.04004","n_code_links":1,"syntology":null},{"paper":"/paper/agglio-global-optimization-for-locally-convex","slug":"agglio-global-optimization-for-locally-convex","title":"AGGLIO: Global Optimization for Locally Convex Functions","date":"2021-11-06","arxiv_id":"2111.03932","n_code_links":1,"syntology":null},{"paper":"/paper/sharp-bounds-for-federated-averaging-local","slug":"sharp-bounds-for-federated-averaging-local","title":"Sharp Bounds for Federated Averaging (Local SGD) and Continuous Perspective","date":"2021-11-05","arxiv_id":"2111.03741","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-fast-parallel-tensor-decomposition-with","title":"A Fast Parallel Tensor Decomposition with Optimal Stochastic Gradient Descent: an Application in Structural Damage Identification","date":"2021-11-04","arxiv_id":"2111.02632","n_code_links":0,"syntology":null},{"paper":null,"slug":"mean-field-analysis-of-piecewise-linear","title":"Mean-field Analysis of Piecewise Linear Solutions for Wide ReLU Networks","date":"2021-11-03","arxiv_id":"2111.02278","n_code_links":0,"syntology":null},{"paper":null,"slug":"regularization-by-misclassification-in-relu","title":"Regularization by Misclassification in ReLU Neural Networks","date":"2021-11-03","arxiv_id":"2111.02154","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-variance-principle-explains-why-dropout-1","title":"Dropout in Training Neural Networks: Flatness of Solution and Noise Structure","date":"2021-11-01","arxiv_id":"2111.01022","n_code_links":0,"syntology":null},{"paper":null,"slug":"predicting-cancer-using-supervised-machine","title":"Predicting Cancer Using Supervised Machine Learning: Mesothelioma","date":"2021-10-31","arxiv_id":"2111.01912","n_code_links":0,"syntology":null}],"record_sha256":"0cddb8829f1fc2059d81cdec1eaef9a292f48e59bb2de23fcdd1003f1809682f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}