{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/adagrad/papers/2","list_of":"/method/adagrad","method":"AdaGrad","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,191],"of":191,"counts":{"archive_papers_tagged":191,"with_a_code_link":70,"where_syntology_ran_a_sample":23,"not_listed_spam_title":0,"listed":191,"listed_where_code_ran":23,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":20,"every_run_a_failure_of_syntologys_instrument":3,"listed_with_a_run_with_no_instrument_failure":20,"listed_every_run_a_failure_of_syntologys_instrument":3,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/adagrad","prev":"/method/adagrad","next":null,"papers":[{"paper":null,"slug":"l2m-practical-posterior-laplace-approximation","title":"L2M: Practical posterior Laplace approximation with optimization-driven second moment estimation","date":"2021-07-09","arxiv_id":"2107.04695","n_code_links":0,"syntology":null},{"paper":null,"slug":"private-adaptive-gradient-methods-for-convex","title":"Private Adaptive Gradient Methods for Convex Optimization","date":"2021-06-25","arxiv_id":"2106.13756","n_code_links":0,"syntology":null},{"paper":"/paper/decreasing-scaling-transition-from-adaptive","slug":"decreasing-scaling-transition-from-adaptive","title":"A decreasing scaling transition scheme from Adam to SGD","date":"2021-06-12","arxiv_id":"2106.06749","n_code_links":2,"syntology":null},{"paper":null,"slug":"comparative-investigation-of-learning","title":"Comparative Investigation of Learning Algorithms for Image Classification with Small Dataset","date":"2021-06-11","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/cross-trajectory-representation-learning-for","slug":"cross-trajectory-representation-learning-for","title":"Cross-Trajectory Representation Learning for Zero-Shot Generalization in RL","date":"2021-06-04","arxiv_id":"2106.02193","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":6,"n_instrument":1,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["bmazoure/ctrl_public"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"local-adaptivity-in-federated-learning","title":"Local Adaptivity in Federated Learning: Convergence and Consistency","date":"2021-06-04","arxiv_id":"2106.02305","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalized-adagrad-g-adagrad-and-adam-a","title":"Generalized AdaGrad (G-AdaGrad) and Adam: A State-Space Perspective","date":"2021-05-31","arxiv_id":"2106.00092","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-relate-depth-and-semantics-for","slug":"learning-to-relate-depth-and-semantics-for","title":"Learning to Relate Depth and Semantics for Unsupervised Domain Adaptation","date":"2021-05-17","arxiv_id":"2105.07830","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["susaha/ctrl-uda"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/ctlr-wic-tsv-target-sense-verification-using","slug":"ctlr-wic-tsv-target-sense-verification-using","title":"CTLR@WiC-TSV: Target Sense Verification using Marked Inputs andPre-trained Models","date":"2021-04-30","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/explainaboard-an-explainable-leaderboard-for","slug":"explainaboard-an-explainable-leaderboard-for","title":"ExplainaBoard: An Explainable Leaderboard for NLP","date":"2021-04-13","arxiv_id":"2104.06387","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploiting-adam-like-optimization-algorithms","title":"Exploiting Adam-like Optimization Algorithms to Improve the Performance of Convolutional Neural Networks","date":"2021-03-26","arxiv_id":"2103.14689","n_code_links":0,"syntology":null},{"paper":null,"slug":"spatio-temporal-neural-network-for-fitting","title":"Spatio-Temporal Neural Network for Fitting and Forecasting COVID-19","date":"2021-03-22","arxiv_id":"2103.11860","n_code_links":0,"syntology":null},{"paper":null,"slug":"categorical-foundations-of-gradient-based","title":"Categorical Foundations of Gradient-Based Learning","date":"2021-03-02","arxiv_id":"2103.01931","n_code_links":0,"syntology":null},{"paper":null,"slug":"variance-reduction-in-training-forecasting","title":"Variance Reduced Training with Stratified Sampling for Forecasting Models","date":"2021-03-02","arxiv_id":"2103.02062","n_code_links":0,"syntology":null},{"paper":null,"slug":"svrg-meets-adagrad-painless-variance","title":"SVRG Meets AdaGrad: Painless Variance Reduction","date":"2021-02-18","arxiv_id":"2102.09645","n_code_links":0,"syntology":null},{"paper":null,"slug":"metagrad-adaptation-using-multiple-learning","title":"MetaGrad: Adaptation using Multiple Learning Rates in Online Learning","date":"2021-02-12","arxiv_id":"2102.06622","n_code_links":0,"syntology":null},{"paper":"/paper/adaptivity-without-compromise-a-momentumized","slug":"adaptivity-without-compromise-a-momentumized","title":"Adaptivity without Compromise: A Momentumized, Adaptive, Dual Averaged Gradient Method for Stochastic Optimization","date":"2021-01-26","arxiv_id":"2101.11075","n_code_links":5,"syntology":null},{"paper":null,"slug":"adam-revisited-a-weighted-past-gradients","title":"Adam revisited: a weighted past gradients perspective","date":"2021-01-01","arxiv_id":"2101.00238","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-gradient-methods-can-be-provably-1","title":"Adaptive Gradient Methods Can Be Provably Faster than SGD with Random Shuffling","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"convergent-adaptive-gradient-methods-in","title":"Convergent Adaptive Gradient Methods in Decentralized Optimization","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/cada-communication-adaptive-distributed-adam","slug":"cada-communication-adaptive-distributed-adam","title":"CADA: Communication-Adaptive Distributed Adam","date":"2020-12-31","arxiv_id":"2012.15469","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":null}},{"paper":null,"slug":"variance-reduction-on-adaptive-stochastic","title":"Variance Reduction on General Adaptive Stochastic Mirror Descent","date":"2020-12-26","arxiv_id":"2012.13760","n_code_links":0,"syntology":null},{"paper":"/paper/the-implicit-bias-for-adaptive-optimization","slug":"the-implicit-bias-for-adaptive-optimization","title":"The Implicit Bias for Adaptive Optimization Algorithms on Homogeneous Neural Networks","date":"2020-12-11","arxiv_id":"2012.06244","n_code_links":1,"syntology":null},{"paper":null,"slug":"asymptotic-study-of-stochastic-adaptive","title":"Asymptotic study of stochastic adaptive algorithm in non-convex landscape","date":"2020-12-10","arxiv_id":"2012.05640","n_code_links":0,"syntology":null},{"paper":null,"slug":"better-full-matrix-regret-via-parameter-free","title":"Better Full-Matrix Regret via Parameter-Free Online Learning","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-better-generalization-of-adaptive","title":"Towards Better Generalization of Adaptive Gradient Methods","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sequential-convergence-of-adagrad-algorithm","title":"Sequential convergence of AdaGrad algorithm for smooth convex optimization","date":"2020-11-24","arxiv_id":"2011.12341","n_code_links":0,"syntology":null},{"paper":null,"slug":"which-minimizer-does-my-neural-network","title":"Which Minimizer Does My Neural Network Converge To?","date":"2020-11-04","arxiv_id":"2011.02408","n_code_links":0,"syntology":null},{"paper":null,"slug":"why-are-convolutional-nets-more-sample-1","title":"Why Are Convolutional Nets More Sample-Efficient than Fully-Connected Nets?","date":"2020-10-16","arxiv_id":"2010.08515","n_code_links":0,"syntology":null},{"paper":null,"slug":"reparametrizing-gradient-descent","title":"Reparametrizing gradient descent","date":"2020-10-09","arxiv_id":"2010.04786","n_code_links":0,"syntology":null},{"paper":null,"slug":"dimension-independence-in-unconstrained","title":"Fast Dimension Independent Private AdaGrad on Publicly Estimated Subspaces","date":"2020-08-14","arxiv_id":"2008.06570","n_code_links":0,"syntology":null},{"paper":"/paper/axiom-based-grad-cam-towards-accurate","slug":"axiom-based-grad-cam-towards-accurate","title":"Axiom-based Grad-CAM: Towards Accurate Visualization and Explanation of CNNs","date":"2020-08-05","arxiv_id":"2008.02312","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Fu0511/XGrad-CAM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-high-probability-analysis-of-adaptive-sgd","title":"A High Probability Analysis of Adaptive SGD with Momentum","date":"2020-07-28","arxiv_id":"2007.14294","n_code_links":0,"syntology":null},{"paper":"/paper/corner-proposal-network-for-anchor-free-two","slug":"corner-proposal-network-for-anchor-free-two","title":"Corner Proposal Network for Anchor-free, Two-stage Object Detection","date":"2020-07-27","arxiv_id":"2007.13816","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-gradient-methods-for-constrained","title":"Adaptive Gradient Methods for Constrained Convex Optimization and Variational Inequalities","date":"2020-07-17","arxiv_id":"2007.08840","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-gradient-methods-can-be-provably","title":"Adaptive Gradient Methods Can Be Provably Faster than SGD after Finite Epochs","date":"2020-06-12","arxiv_id":"2006.07037","n_code_links":0,"syntology":null},{"paper":"/paper/adaptive-gradient-methods-converge-faster","slug":"adaptive-gradient-methods-converge-faster","title":"Adaptive Gradient Methods Converge Faster with Over-Parameterization (but you should do a line-search)","date":"2020-06-11","arxiv_id":"2006.06835","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/adahessian-an-adaptive-second-order-optimizer","slug":"adahessian-an-adaptive-second-order-optimizer","title":"ADAHESSIAN: An Adaptive Second Order Optimizer for Machine Learning","date":"2020-06-01","arxiv_id":"2006.00719","n_code_links":4,"syntology":{"ran":5,"of":11,"n_ran_checked":4,"n_instrument":1,"unverified":6,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["amirgholami/adahessian"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"on-the-convergence-of-adam-and-adagrad","title":"A Simple Convergence Proof of Adam and Adagrad","date":"2020-03-05","arxiv_id":"2003.02395","n_code_links":0,"syntology":null},{"paper":null,"slug":"stagewise-enlargement-of-batch-size-for-sgd","title":"Stagewise Enlargement of Batch Size for SGD-based Learning","date":"2020-02-26","arxiv_id":"2002.11601","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-online-learning-with-varying-norms","title":"Adaptive Online Learning with Varying Norms","date":"2020-02-10","arxiv_id":"2002.03963","n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-the-generalization-of-adaptive","title":"Revisiting the Generalization of Adaptive Gradient Methods","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-better-understanding-of-adaptive-1","title":"Towards Better Understanding of Adaptive Gradient Algorithms in Generative Adversarial Nets","date":"2019-12-26","arxiv_id":"1912.11940","n_code_links":0,"syntology":null},{"paper":null,"slug":"second-order-information-in-first-order","title":"Second-order Information in First-order Optimization Methods","date":"2019-12-20","arxiv_id":"1912.09926","n_code_links":0,"syntology":null},{"paper":"/paper/parameter-continuation-methods-for-the","slug":"parameter-continuation-methods-for-the","title":"Parameter Continuation Methods for the Optimization of Deep Neural Networks","date":"2019-12-16","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/memory-efficient-adaptive-optimization","slug":"memory-efficient-adaptive-optimization","title":"Memory Efficient Adaptive Optimization","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/an-adaptive-and-momental-bound-method-for","slug":"an-adaptive-and-momental-bound-method-for","title":"An Adaptive and Momental Bound Method for Stochastic Learning","date":"2019-10-27","arxiv_id":"1910.12249","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lancopku/AdaMod"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"implementation-of-a-modified-nesterovs","title":"Implementation of a modified Nesterov's Accelerated quasi-Newton Method on Tensorflow","date":"2019-10-21","arxiv_id":"1910.09158","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-step-sizes-in-variance-reduction-via","title":"Adaptive Step Sizes in Variance Reduction via Regularization","date":"2019-10-15","arxiv_id":"1910.06532","n_code_links":0,"syntology":null},{"paper":"/paper/diffgrad-an-optimization-method-for","slug":"diffgrad-an-optimization-method-for","title":"diffGrad: An Optimization Method for Convolutional Neural Networks","date":"2019-09-12","arxiv_id":"1909.11015","n_code_links":1,"syntology":null},{"paper":"/paper/ctrl-a-conditional-transformer-language-model-1","slug":"ctrl-a-conditional-transformer-language-model-1","title":"CTRL: A Conditional Transformer Language Model for Controllable Generation","date":"2019-09-11","arxiv_id":"1909.05858","n_code_links":8,"syntology":{"ran":14,"of":14,"n_ran_checked":11,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"14 ran (of which 3 constructed an object rather than computing a result; 11 with no instrument failure: 3 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"meta-descent-for-online-continual-prediction","title":"Meta-descent for Online, Continual Prediction","date":"2019-07-17","arxiv_id":"1907.07751","n_code_links":0,"syntology":null},{"paper":"/paper/augmenting-self-attention-with-persistent","slug":"augmenting-self-attention-with-persistent","title":"Augmenting Self-attention with Persistent Memory","date":"2019-07-02","arxiv_id":"1907.01470","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/adaptively-preconditioned-stochastic-gradient","slug":"adaptively-preconditioned-stochastic-gradient","title":"Adaptively Preconditioned Stochastic Gradient Langevin Dynamics","date":"2019-06-10","arxiv_id":"1906.04324","n_code_links":1,"syntology":null},{"paper":"/paper/the-implicit-bias-of-adagrad-on-separable","slug":"the-implicit-bias-of-adagrad-on-separable","title":"The Implicit Bias of AdaGrad on Separable Data","date":"2019-06-09","arxiv_id":"1906.03559","n_code_links":1,"syntology":null},{"paper":"/paper/adaoja-adaptive-learning-rates-for-streaming","slug":"adaoja-adaptive-learning-rates-for-streaming","title":"AdaOja: Adaptive Learning Rates for Streaming PCA","date":"2019-05-28","arxiv_id":"1905.12115","n_code_links":1,"syntology":null},{"paper":null,"slug":"hyper-regularization-an-adaptive-choice-for","title":"Hyper-Regularization: An Adaptive Choice for the Learning Rate in Gradient Descent","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/adaptive-gradient-methods-with-dynamic-bound","slug":"adaptive-gradient-methods-with-dynamic-bound","title":"Adaptive Gradient Methods with Dynamic Bound of Learning Rate","date":"2019-02-26","arxiv_id":"1902.09843","n_code_links":5,"syntology":null},{"paper":null,"slug":"global-convergence-of-adaptive-gradient","title":"Global Convergence of Adaptive Gradient Methods for An Over-parameterized Neural Network","date":"2019-02-19","arxiv_id":"1902.07111","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-universal-algorithm-for-variational","title":"A Universal Algorithm for Variational Inequalities Adaptive to Smoothness and Noise","date":"2019-02-05","arxiv_id":"1902.01637","n_code_links":0,"syntology":null},{"paper":"/paper/compressing-gradient-optimizers-via-count","slug":"compressing-gradient-optimizers-via-count","title":"Compressing Gradient Optimizers via Count-Sketches","date":"2019-02-01","arxiv_id":"1902.00179","n_code_links":1,"syntology":null},{"paper":"/paper/memory-efficient-adaptive-optimization-for","slug":"memory-efficient-adaptive-optimization-for","title":"Memory-Efficient Adaptive Optimization","date":"2019-01-30","arxiv_id":"1901.11150","n_code_links":4,"syntology":null},{"paper":null,"slug":"a-sufficient-condition-for-convergences-of","title":"A Sufficient Condition for Convergences of Adam and RMSProp","date":"2018-11-23","arxiv_id":"1811.09358","n_code_links":0,"syntology":null},{"paper":"/paper/practical-bayesian-learning-of-neural","slug":"practical-bayesian-learning-of-neural","title":"Practical Bayesian Learning of Neural Networks via Adaptive Optimisation Methods","date":"2018-11-08","arxiv_id":"1811.03679","n_code_links":1,"syntology":null},{"paper":"/paper/riemannian-adaptive-optimization-methods","slug":"riemannian-adaptive-optimization-methods","title":"Riemannian Adaptive Optimization Methods","date":"2018-10-01","arxiv_id":"1810.00760","n_code_links":1,"syntology":null},{"paper":null,"slug":"universal-stagewise-learning-for-non-convex","title":"Universal Stagewise Learning for Non-Convex Problems with Convergence on Averaged Solutions","date":"2018-08-20","arxiv_id":"1808.06296","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-weighted-adagrad-with","title":"A Unified Analysis of AdaGrad with Weighted Aggregation and Momentum Acceleration","date":"2018-08-10","arxiv_id":"1808.03408","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-a-class-of-adam-type","title":"On the Convergence of A Class of Adam-Type Algorithms for Non-Convex Optimization","date":"2018-08-08","arxiv_id":"1808.02941","n_code_links":0,"syntology":null},{"paper":null,"slug":"sadagrad-strongly-adaptive-stochastic","title":"SADAGRAD: Strongly Adaptive Stochastic Gradient Methods","date":"2018-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/adagrad-stepsizes-sharp-convergence-over","slug":"adagrad-stepsizes-sharp-convergence-over","title":"AdaGrad stepsizes: Sharp convergence over nonconvex landscapes","date":"2018-06-05","arxiv_id":"1806.01811","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-stochastic-gradient-1","title":"On the Convergence of Stochastic Gradient Descent with Adaptive Stepsizes","date":"2018-05-21","arxiv_id":"1805.08114","n_code_links":0,"syntology":null},{"paper":null,"slug":"block-mean-approximation-for-efficient-second","title":"Block Mean Approximation for Efficient Second Order Optimization","date":"2018-04-16","arxiv_id":"1804.05484","n_code_links":0,"syntology":null},{"paper":"/paper/shampoo-preconditioned-stochastic-tensor","slug":"shampoo-preconditioned-stochastic-tensor","title":"Shampoo: Preconditioned Stochastic Tensor Optimization","date":"2018-02-26","arxiv_id":"1802.09568","n_code_links":3,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"lsh-sampling-breaks-the-computational-chicken","title":"LSH-SAMPLING BREAKS THE COMPUTATIONAL CHICKEN-AND-EGG LOOP IN ADAPTIVE STOCHASTIC GRADIENT ESTIMATION","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/improving-generalization-performance-by","slug":"improving-generalization-performance-by","title":"Improving Generalization Performance by Switching from Adam to SGD","date":"2017-12-20","arxiv_id":"1712.07628","n_code_links":6,"syntology":null},{"paper":null,"slug":"adabatch-efficient-gradient-aggregation-rules","title":"AdaBatch: Efficient Gradient Aggregation Rules for Sequential and Parallel Stochastic Gradient Methods","date":"2017-11-06","arxiv_id":"1711.01761","n_code_links":0,"syntology":null},{"paper":null,"slug":"why-adagrad-fails-for-online-topic-modeling","title":"Why ADAGRAD Fails for Online Topic Modeling","date":"2017-09-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-unified-approach-to-adaptive-regularization","title":"A Unified Approach to Adaptive Regularization in Online and Stochastic Optimization","date":"2017-06-20","arxiv_id":"1706.06569","n_code_links":0,"syntology":null},{"paper":"/paper/yellowfin-and-the-art-of-momentum-tuning","slug":"yellowfin-and-the-art-of-momentum-tuning","title":"YellowFin and the Art of Momentum Tuning","date":"2017-06-12","arxiv_id":"1706.03471","n_code_links":2,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":null}},{"paper":"/paper/the-marginal-value-of-adaptive-gradient","slug":"the-marginal-value-of-adaptive-gradient","title":"The Marginal Value of Adaptive Gradient Methods in Machine Learning","date":"2017-05-23","arxiv_id":"1705.08292","n_code_links":3,"syntology":null},{"paper":"/paper/efficient-parallel-translating-embedding-for","slug":"efficient-parallel-translating-embedding-for","title":"Efficient Parallel Translating Embedding For Knowledge Graphs","date":"2017-03-30","arxiv_id":"1703.10316","n_code_links":1,"syntology":null},{"paper":"/paper/wasserstein-gan","slug":"wasserstein-gan","title":"Wasserstein GAN","date":"2017-01-26","arxiv_id":"1701.07875","n_code_links":120,"syntology":{"ran":16,"of":22,"n_ran_checked":4,"n_instrument":12,"unverified":6,"pointer_only":13,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 12 where Syntology's instrument failed) · 6 unverified","official":null}},{"paper":"/paper/improving-neural-language-models-with-a","slug":"improving-neural-language-models-with-a","title":"Improving Neural Language Models with a Continuous Cache","date":"2016-12-13","arxiv_id":"1612.04426","n_code_links":14,"syntology":null},{"paper":null,"slug":"scalable-adaptive-stochastic-optimization","title":"Scalable Adaptive Stochastic Optimization Using Random Projections","date":"2016-11-21","arxiv_id":"1611.06652","n_code_links":0,"syntology":null},{"paper":null,"slug":"relativistic-monte-carlo","title":"Relativistic Monte Carlo","date":"2016-09-14","arxiv_id":"1609.04388","n_code_links":0,"syntology":null},{"paper":null,"slug":"compadagrad-a-compressed-complementary","title":"CompAdaGrad: A Compressed, Complementary, Computationally-Efficient Adaptive Gradient Method","date":"2016-09-12","arxiv_id":"1609.03319","n_code_links":0,"syntology":null},{"paper":"/paper/bridging-the-gap-between-stochastic-gradient","slug":"bridging-the-gap-between-stochastic-gradient","title":"Bridging the Gap between Stochastic Gradient MCMC and Stochastic Optimization","date":"2015-12-25","arxiv_id":"1512.07962","n_code_links":1,"syntology":null},{"paper":null,"slug":"speed-learning-on-the-fly","title":"Speed learning on the fly","date":"2015-11-08","arxiv_id":"1511.02540","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaqn-an-adaptive-quasi-newton-algorithm-for","title":"adaQN: An Adaptive Quasi-Newton Algorithm for Training RNNs","date":"2015-11-04","arxiv_id":"1511.01169","n_code_links":0,"syntology":null},{"paper":"/paper/path-sgd-path-normalized-optimization-in-deep","slug":"path-sgd-path-normalized-optimization-in-deep","title":"Path-SGD: Path-Normalized Optimization in Deep Neural Networks","date":"2015-06-08","arxiv_id":"1506.02617","n_code_links":1,"syntology":null},{"paper":null,"slug":"dropout-training-as-adaptive-regularization","title":"Dropout Training as Adaptive Regularization","date":"2013-07-04","arxiv_id":"1307.1493","n_code_links":0,"syntology":null}],"record_sha256":"73f01a3fe0703d9977f19e04638406170aa4c5130d917e0e12fb9918ba7dafe2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}