{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T16:31:54Z","timestamp":1783787514395,"version":"3.55.0"},"reference-count":27,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2018,5,1]],"date-time":"2018-05-01T00:00:00Z","timestamp":1525132800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2018,5]]},"DOI":"10.1109\/tnnls.2017.2672978","type":"journal-article","created":{"date-parts":[[2017,3,10]],"date-time":"2017-03-10T07:33:13Z","timestamp":1489131193000},"page":"1454-1466","source":"Crossref","is-referenced-by-count":84,"title":["Preconditioned Stochastic Gradient Descent"],"prefix":"10.1109","volume":"29","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3853-2702","authenticated-orcid":false,"given":"Xi-Lin","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1137\/140954362"},{"key":"ref11","first-page":"1737","article-title":"SGD-QN: Careful quasi-Newton stochastic gradient descent","volume":"10","author":"antoine","year":"2009","journal-title":"J Mach Learn Res"},{"key":"ref12","author":"hinton","year":"2016","journal-title":"Neural Networks for Machine Learning"},{"key":"ref13","first-page":"1139","article-title":"On the importance of momentum and initialization in deep learning","author":"sutskever","year":"2013","journal-title":"Proc 30th Int Conf Mach Learn"},{"key":"ref14","author":"schaul","year":"2012","journal-title":"No More Pesky Learning Rates"},{"key":"ref15","first-page":"1504","article-title":"Equilibrated adaptive learning rates for non-convex optimization","author":"dauphin","year":"2015","journal-title":"Proc 28th Int Conf Adv Neural Inf Process Syst"},{"key":"ref16","first-page":"1788","article-title":"Preconditioned stochastic gradient Langevin dynamics for deep neural networks","author":"li","year":"2016","journal-title":"Proc 30th AAAI Conf Artif Intell"},{"key":"ref17","first-page":"2970","article-title":"Preconditioned spectral descent for deep learning","author":"carlson","year":"2015","journal-title":"Proc 28th Int Conf Neural Inf Process Syst"},{"key":"ref18","article-title":"Parallel training of DNNs with natural gradient and parameter averaging","author":"povey","year":"2015","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref19","first-page":"2408","article-title":"Optimizing neural networks with kronecker-factored approximate curvature","author":"martens","year":"2015","journal-title":"Proceedings of the 32nd Intl Conf on Machine Learning"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/5.58337"},{"key":"ref27","first-page":"265","article-title":"On the algorithmic implementation of multiclass kernel-based vector machines","volume":"2","author":"crammer","year":"2001","journal-title":"J Mach Learn Res"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1038\/323533a0"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/72.286919"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/5.726791"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-35289-8_27"},{"key":"ref7","author":"demuth","year":"2002","journal-title":"Neural Network Toolbox for Use with MATLAB"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TCOM.1980.1094608"},{"key":"ref9","first-page":"436","article-title":"A stochastic quasi-Newton method for online convex optimization","volume":"2","author":"schraudolph","year":"2007","journal-title":"J Mach Learn Res"},{"key":"ref1","author":"widrow","year":"1985","journal-title":"Adaptive Signal Processing"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/78.553476"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/78.414774"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1162\/089976698300017746"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/72.279181"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref26","author":"lecun","year":"2016","journal-title":"The MNIST database"},{"key":"ref25","author":"li","year":"2016","journal-title":"Recurrent neural network training with preconditioned stochastic gradient descent"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5962385\/8338465\/07875097.pdf?arnumber=7875097","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,12]],"date-time":"2022-01-12T16:22:53Z","timestamp":1642004573000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/7875097\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,5]]},"references-count":27,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2017.2672978","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018,5]]}}}