{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/second-order-methods/papers/2","list_of":"/task/second-order-methods","task":"Second-order methods","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,181],"of":181,"counts":{"archive_papers_tagged":181,"with_a_code_link":52,"where_syntology_ran_a_sample":12,"not_listed_spam_title":0,"listed":181,"listed_where_code_ran":12,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":11,"every_run_a_failure_of_syntologys_instrument":1,"listed_with_a_run_with_no_instrument_failure":11,"listed_every_run_a_failure_of_syntologys_instrument":1,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/second-order-methods","prev":"/task/second-order-methods","next":null,"papers":[{"url":null,"slug":"on-the-efficiency-of-stochastic-quasi-newton","title":"On the efficiency of Stochastic Quasi-Newton Methods for Deep Learning","date":"2022-05-18","arxiv_id":"2205.09121","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-fast-exact-subproblem-solver-for","title":"A Novel Fast Exact Subproblem Solver for Stochastic Quasi-Newton Cubic Regularized Optimization","date":"2022-04-19","arxiv_id":"2204.09116","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-stochastic-probabilistic","title":"Accelerating Stochastic Probabilistic Inference","date":"2022-03-15","arxiv_id":"2203.07585","repositories_listed":0,"syntology":null},{"url":null,"slug":"amortized-proximal-optimization","title":"Amortized Proximal Optimization","date":"2022-02-28","arxiv_id":"2203.00089","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mini-block-natural-gradient-method-for-deep","title":"A Mini-Block Fisher Method for Deep Neural Networks","date":"2022-02-08","arxiv_id":"2202.04124","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerated-projected-gradient-method-for-the","title":"Accelerated Projected Gradient Method for the Optimization of Cell-Free Massive MIMO Downlink","date":"2022-01-12","arxiv_id":"2201.04440","repositories_listed":0,"syntology":null},{"url":null,"slug":"sc-reg-training-overparameterized-neural","title":"SCORE: Approximating Curvature Information under Self-Concordant Regularization","date":"2021-12-14","arxiv_id":"2112.07344","repositories_listed":0,"syntology":null},{"url":null,"slug":"newton-methods-based-convolution-neural","title":"Newton methods based convolution neural networks using parallel processing","date":"2021-12-02","arxiv_id":"2112.01401","repositories_listed":0,"syntology":null},{"url":null,"slug":"basis-matters-better-communication-efficient","title":"Basis Matters: Better Communication-Efficient Second Order Methods for Federated Learning","date":"2021-11-02","arxiv_id":"2111.01847","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-regret-bounds-for-deep-online-1","title":"Provable Regret Bounds for Deep Online Learning and Control","date":"2021-10-15","arxiv_id":"2110.07807","repositories_listed":0,"syntology":null},{"url":null,"slug":"kkt-conditions-first-order-and-second-order","title":"KKT Conditions, First-Order and Second-Order Optimization, and Distributed Optimization: Tutorial and Survey","date":"2021-10-05","arxiv_id":"2110.01858","repositories_listed":0,"syntology":null},{"url":null,"slug":"slim-qn-a-stochastic-light-momentumized-quasi","title":"SLIM-QN: A Stochastic, Light, Momentumized Quasi-Newton Optimizer for Deep Neural Networks","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"doubly-adaptive-scaled-algorithm-for-machine","title":"Doubly Adaptive Scaled Algorithm for Machine Learning Using Second-Order Information","date":"2021-09-11","arxiv_id":"2109.05198","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-second-order-methods-via-natural","title":"Structured second-order methods via natural gradient descent","date":"2021-07-22","arxiv_id":"2107.10884","repositories_listed":0,"syntology":null},{"url":null,"slug":"bilinear-parameterization-for-non-separable","title":"Bilinear Parameterization for Non-Separable Singular Value Penalties","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fednl-making-newton-type-methods-applicable","title":"FedNL: Making Newton-Type Methods Applicable to Federated Learning","date":"2021-06-05","arxiv_id":"2106.02969","repositories_listed":0,"syntology":null},{"url":null,"slug":"exact-stochastic-second-order-deep-learning","title":"Exact Stochastic Second Order Deep Learning","date":"2021-04-08","arxiv_id":"2104.03804","repositories_listed":0,"syntology":null},{"url":null,"slug":"quasi-newton-quasi-monte-carlo-for","title":"Quasi-Newton Quasi-Monte Carlo for variational Bayes","date":"2021-04-07","arxiv_id":"2104.02865","repositories_listed":0,"syntology":null},{"url":null,"slug":"research-of-damped-newton-stochastic-gradient","title":"Research of Damped Newton Stochastic Gradient Descent Method for Neural Network Training","date":"2021-03-31","arxiv_id":"2103.16764","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-distributed-optimisation-framework","title":"A Distributed Optimisation Framework Combining Natural Gradient with Hessian-Free for Discriminative Sequence Training","date":"2021-03-12","arxiv_id":"2103.07554","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-augmented-sketches-for-hessians-1","title":"Learning-Augmented Sketches for Hessians","date":"2021-02-24","arxiv_id":"2102.12317","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-second-order-methods-with-fast","title":"Distributed Second Order Methods with Fast Rates and Compressed Communication","date":"2021-02-14","arxiv_id":"2102.07158","repositories_listed":0,"syntology":null},{"url":null,"slug":"kronecker-factored-quasi-newton-methods-for","title":"Kronecker-factored Quasi-Newton Methods for Deep Learning","date":"2021-02-12","arxiv_id":"2102.06737","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-chaos-theory-approach-to-understand-neural","title":"A Chaos Theory Approach to Understand Neural Network Optimization","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-single-pass-stochastic-gradient","title":"Adaptive Single-Pass Stochastic Gradient Descent in Input Sparsity Time","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-gradient-descent-with-large","title":"Noise and Fluctuation of Finite Learning Rate Stochastic Gradient Descent","date":"2020-12-07","arxiv_id":"2012.03636","repositories_listed":0,"syntology":null},{"url":null,"slug":"second-order-neural-network-training-using","title":"Second-order Neural Network Training Using Complex-step Directional Derivative","date":"2020-09-15","arxiv_id":"2009.07098","repositories_listed":0,"syntology":null},{"url":null,"slug":"utility-maximization-for-large-scale-cell","title":"Utility Maximization for Large-Scale Cell-Free Massive MIMO Downlink","date":"2020-09-15","arxiv_id":"2009.07167","repositories_listed":0,"syntology":null},{"url":null,"slug":"debiasing-distributed-second-order","title":"Debiasing Distributed Second Order Optimization with Surrogate Sketching and Scaled Regularization","date":"2020-07-02","arxiv_id":"2007.01327","repositories_listed":0,"syntology":null},{"url":null,"slug":"second-order-information-in-non-convex","title":"Second-Order Information in Non-Convex Stochastic Optimization: Power and Limitations","date":"2020-06-24","arxiv_id":"2006.13476","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-does-preconditioning-help-or-hurt","title":"When Does Preconditioning Help or Hurt Generalization?","date":"2020-06-18","arxiv_id":"2006.10732","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-block-coordinate-descent-optimizer-for","title":"A block coordinate descent optimizer for classification problems exploiting convexity","date":"2020-06-17","arxiv_id":"2006.10123","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-stochastic-quasi-newton-methods","title":"Enhance Curvature Information by Structured Stochastic Quasi-Newton Methods","date":"2020-06-17","arxiv_id":"2006.09606","repositories_listed":0,"syntology":null},{"url":null,"slug":"sonia-a-symmetric-blockwise-truncated","title":"SONIA: A Symmetric Blockwise Truncated Optimization Algorithm","date":"2020-06-06","arxiv_id":"2006.03949","repositories_listed":0,"syntology":null},{"url":null,"slug":"critical-point-finding-methods-reveal","title":"Critical Point-Finding Methods Reveal Gradient-Flat Regions of Deep Network Losses","date":"2020-03-23","arxiv_id":"2003.10397","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-subspace-cubic-newton-method","title":"Stochastic Subspace Cubic Newton Method","date":"2020-02-21","arxiv_id":"2002.09526","repositories_listed":0,"syntology":null},{"url":null,"slug":"differential-dynamic-programming-neural","title":"DDPNOpt: Differential Dynamic Programming Neural Optimizer","date":"2020-02-20","arxiv_id":"2002.08809","repositories_listed":0,"syntology":null},{"url":null,"slug":"curvature-corrected-learning-dynamics-in-deep","title":"Curvature-corrected learning dynamics in deep neural networks","date":"2020-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-model-based-policy-optimization","title":"Hierarchical model-based policy optimization: from actions to action sequences and back","date":"2019-11-28","arxiv_id":"1912.01448","repositories_listed":0,"syntology":null},{"url":null,"slug":"implementation-of-a-modified-nesterovs","title":"Implementation of a modified Nesterov's Accelerated quasi-Newton Method on Tensorflow","date":"2019-10-21","arxiv_id":"1910.09158","repositories_listed":0,"syntology":null},{"url":null,"slug":"exact-analysis-of-curvature-corrected","title":"EXACT ANALYSIS OF CURVATURE CORRECTED LEARNING DYNAMICS IN DEEP LINEAR NETWORKS","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"quasi-newton-optimization-methods-for-deep","title":"Quasi-Newton Optimization Methods For Deep Learning Applications","date":"2019-09-04","arxiv_id":"1909.01994","repositories_listed":0,"syntology":null},{"url":null,"slug":"practical-newton-type-distributed-learning","title":"Practical Newton-Type Distributed Learning using Gradient Based Approximations","date":"2019-07-22","arxiv_id":"1907.09562","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-descent-for-online-continual-prediction","title":"Meta-descent for Online, Continual Prediction","date":"2019-07-17","arxiv_id":"1907.07751","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-gram-gauss-newton-method-learning","title":"Gram-Gauss-Newton Method: Learning Overparameterized Neural Networks for Regression Problems","date":"2019-05-28","arxiv_id":"1905.11675","repositories_listed":0,"syntology":null},{"url":null,"slug":"bilinear-parameterization-for-differentiable","title":"Bilinear Parameterization For Differentiable Rank-Regularization","date":"2018-11-27","arxiv_id":"1811.11088","repositories_listed":0,"syntology":null},{"url":null,"slug":"quasi-newton-optimization-in-deep-q-learning","title":"Deep Reinforcement Learning via L-BFGS Optimization","date":"2018-11-06","arxiv_id":"1811.02693","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-second-order-methods-for-non","title":"Stochastic Second-order Methods for Non-convex Optimization with Inexact Hessian and Gradient","date":"2018-09-26","arxiv_id":"1809.09853","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-distributed-second-order-algorithm-you-can","title":"A Distributed Second-Order Algorithm You Can Trust","date":"2018-06-20","arxiv_id":"1806.07569","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpu-accelerated-sub-sampled-newtons-method","title":"GPU Accelerated Sub-Sampled Newton's Method","date":"2018-02-26","arxiv_id":"1802.09113","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-many-faces-of-exponential-weights-in","title":"The Many Faces of Exponential Weights in Online Learning","date":"2018-02-21","arxiv_id":"1802.07543","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-second-order-methods-for-deep","title":"A comparison of second-order methods for deep convolutional neural networks","date":"2018-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"block-diagonal-hessian-free-optimization-for","title":"Block-diagonal Hessian-free Optimization for Training Neural Networks","date":"2017-12-20","arxiv_id":"1712.07296","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-second-order-online-kernel-learning","title":"Efficient Second-Order Online Kernel Learning with Adaptive Embedding","date":"2017-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nesterovs-acceleration-for-approximate-newton","title":"Nesterov's Acceleration For Approximate Newton","date":"2017-10-17","arxiv_id":"1710.08496","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-sgd-for-distributed-deep","title":"Accelerating SGD for Distributed Deep-Learning Using Approximated Hessian Matrix","date":"2017-09-15","arxiv_id":"1709.05069","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-generic-approach-for-escaping-saddle-points","title":"A Generic Approach for Escaping Saddle points","date":"2017-09-05","arxiv_id":"1709.01434","repositories_listed":0,"syntology":null},{"url":null,"slug":"first-and-second-order-methods-for-online","title":"First and Second Order Methods for Online Convolutional Dictionary Learning","date":"2017-08-31","arxiv_id":"1709.00106","repositories_listed":0,"syntology":null},{"url":null,"slug":"second-order-optimization-for-non-convex","title":"Second-Order Optimization for Non-Convex Machine Learning: An Empirical Study","date":"2017-08-25","arxiv_id":"1708.07827","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-newton-methods-and-their-local","title":"Approximate Newton Methods and Their Local Convergence","date":"2017-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"second-order-kernel-online-convex","title":"Second-Order Kernel Online Convex Optimization with Adaptive Sketching","date":"2017-06-15","arxiv_id":"1706.04892","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-empirical-risk-minimization-via","title":"Large Scale Empirical Risk Minimization via Truncated Adaptive Newton Method","date":"2017-05-22","arxiv_id":"1705.07957","repositories_listed":0,"syntology":null},{"url":null,"slug":"nestrovs-acceleration-for-second-order-method","title":"Nestrov's Acceleration For Second Order Method","date":"2017-05-19","arxiv_id":"1705.07171","repositories_listed":0,"syntology":null},{"url":null,"slug":"biologically-inspired-protection-of-deep","title":"Biologically inspired protection of deep networks from adversarial attacks","date":"2017-03-27","arxiv_id":"1703.09202","repositories_listed":0,"syntology":null},{"url":null,"slug":"highly-efficient-hierarchical-online","title":"Highly Efficient Hierarchical Online Nonlinear Regression Using Second Order Methods","date":"2017-01-18","arxiv_id":"1701.05053","repositories_listed":0,"syntology":null},{"url":null,"slug":"oracle-complexity-of-second-order-methods-for","title":"Oracle Complexity of Second-Order Methods for Finite-Sum Problems","date":"2016-11-15","arxiv_id":"1611.04982","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-heavy-ball","title":"Stochastic Heavy Ball","date":"2016-09-14","arxiv_id":"1609.04228","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-sampled-newton-methods-with-non-uniform","title":"Sub-sampled Newton Methods with Non-uniform Sampling","date":"2016-07-02","arxiv_id":"1607.00559","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-decentralized-quasi-newton-method-for-dual","title":"A Decentralized Quasi-Newton Method for Dual Formulations of Consensus Optimization","date":"2016-03-23","arxiv_id":"1603.07195","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-quasi-newton-langevin-monte-carlo","title":"Stochastic Quasi-Newton Langevin Monte Carlo","date":"2016-02-10","arxiv_id":"1602.03442","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-sampled-newton-methods-ii-local","title":"Sub-Sampled Newton Methods II: Local Convergence Rates","date":"2016-01-18","arxiv_id":"1601.04738","repositories_listed":0,"syntology":null},{"url":null,"slug":"newton-stein-method-a-second-order-method-for","title":"Newton-Stein Method: A Second Order Method for GLMs via Stein's Lemma","date":"2015-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"alternating-direction-method-of-multipliers","title":"Alternating direction method of multipliers for regularized multiclass support vector machines","date":"2015-11-30","arxiv_id":"1511.09153","repositories_listed":0,"syntology":null},{"url":null,"slug":"newton-stein-method-an-optimization-method","title":"Newton-Stein Method: An optimization method for GLMs via Stein's Lemma","date":"2015-11-28","arxiv_id":"1511.08895","repositories_listed":0,"syntology":null},{"url":null,"slug":"saddle-free-hessian-free-optimization","title":"Saddle-free Hessian-free Optimization","date":"2015-05-30","arxiv_id":"1506.00059","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-flexible-tensor-block-coordinate-ascent","title":"A Flexible Tensor Block Coordinate Ascent Scheme for Hypergraph Matching","date":"2015-04-29","arxiv_id":"1504.07907","repositories_listed":0,"syntology":null},{"url":null,"slug":"quic-dirty-a-quadratic-approximation-approach","title":"QUIC & DIRTY: A Quadratic Approximation Approach for Dirty Statistical Models","date":"2014-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-scaled-gradient-projection-method-for","title":"A scaled gradient projection method for Bayesian learning in dynamical systems","date":"2014-06-25","arxiv_id":"1406.6603","repositories_listed":0,"syntology":null},{"url":null,"slug":"res-regularized-stochastic-bfgs-algorithm","title":"RES: Regularized Stochastic BFGS Algorithm","date":"2014-01-29","arxiv_id":"1401.7625","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-importance-of-initialization-and","title":"On the importance of initialization and momentum in deep learning","date":"2013-05-26","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"second-order-bilinear-discriminant-analysis","title":"Second Order Bilinear Discriminant Analysis for single trial EEG analysis","date":"2007-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"72913de1247dd010b66c15b7728c7a24e6984108fb76c7b5351c1db134d34197","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}