{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T20:24:52Z","timestamp":1784233492396,"version":"3.55.0"},"reference-count":23,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2021,4,8]],"date-time":"2021-04-08T00:00:00Z","timestamp":1617840000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,4,8]],"date-time":"2021-04-08T00:00:00Z","timestamp":1617840000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2021,5]]},"DOI":"10.1007\/s11432-020-3163-0","type":"journal-article","created":{"date-parts":[[2021,4,12]],"date-time":"2021-04-12T13:02:53Z","timestamp":1618232573000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":20,"title":["Learning dynamics of gradient descent optimization in deep neural networks"],"prefix":"10.1007","volume":"64","author":[{"given":"Wei","family":"Wu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoyuan","family":"Jing","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wencai","family":"Du","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guoliang","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,4,8]]},"reference":[{"key":"3163_CR1","unstructured":"Ruder S. An overview of gradient descent optimization algorithms. 2016. ArXiv:1609.04747"},{"key":"3163_CR2","doi-asserted-by":"crossref","unstructured":"An W P, Wang H Q, Sun Q Y, et al. A PID controller approach for stochastic optimization of deep networks. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, 2018. 8522\u20138531","DOI":"10.1109\/CVPR.2018.00889"},{"key":"3163_CR3","doi-asserted-by":"crossref","unstructured":"Kim D, Kim J, Kwon J, et al. Depth-controllable very deep super-resolution network. In: Proceedings of International Joint Conference on Neural Networks, 2019. 1\u20138","DOI":"10.1109\/IJCNN.2019.8851874"},{"key":"3163_CR4","unstructured":"Hinton G, Srivastava N, Swersky K. Overview of mini-batch gradient descent. 2012. http:\/\/www.cs.toronto.edu\/~tijmen\/csc321\/slides\/lecture_slides_lec6.pdf"},{"key":"3163_CR5","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1016\/S0893-6080(98)00116-6","volume":"12","author":"N Qian","year":"1999","unstructured":"Qian N. On the momentum term in gradient descent learning algorithms. Neural Netw, 1999, 12: 145\u2013151","journal-title":"Neural Netw"},{"key":"3163_CR6","first-page":"2121","volume":"12","author":"J Duchi","year":"2011","unstructured":"Duchi J, Hazan E, Singer Y. Adaptive subgradient methods for online learning and stochastic optimization. J Mach Learn Res, 2011, 12: 2121\u20132159","journal-title":"J Mach Learn Res"},{"key":"3163_CR7","unstructured":"Zeiler M D. Adadelta: an adaptive learning rate method. 2012. ArXiv:1212.5701"},{"key":"3163_CR8","unstructured":"Dauphin Y N, de Vries H, Bengio Y. Equilibrated adaptive learning rates for nonconvex optimization. In: Proceedings of Conference and Workshop on Neural Information Processing Systems, 2015"},{"key":"3163_CR9","unstructured":"Kingma D, Ba J. Adam: a method for stochastic optimization. In: Proceedings of International Conference on Learning Representations, 2015. 1\u201315"},{"key":"3163_CR10","unstructured":"Reddi S J, Kale S, Kumar S. On the convergence of ADAM and beyond. In: Proceedings of International Conference on Learning Representations, 2018. 1\u201323"},{"key":"3163_CR11","unstructured":"Luo L C, Xiong Y H, Liu Y, et al. Adaptive gradient methods with dynamic bound of learning rate. In: Proceedings of International Conference on Learning Representations, 2019. 1\u201319"},{"key":"3163_CR12","unstructured":"Saxe A M, McClelland J L, Ganguli S. Exact solutions to the nonlinear dynamics of learning in deep linear neural networks. 2013. ArXiv:1312.6120"},{"key":"3163_CR13","doi-asserted-by":"publisher","first-page":"4238","DOI":"10.1109\/TNNLS.2017.2760979","volume":"29","author":"T H Lee","year":"2018","unstructured":"Lee T H, Trinh H M, Park J H. Stability analysis of neural networks with time-varying delay by constructing novel Lyapunov functionals. IEEE Trans Neural Netw Learn Syst, 2018, 29: 4238\u20134247","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"3163_CR14","doi-asserted-by":"crossref","unstructured":"Faydasicok O, Arik S. A novel criterion for global asymptotic stability of neutral type neural networks with discrete time delays. In: Proceedings of International Conference on Neural Information Processing, 2018. 353\u2013360","DOI":"10.1007\/978-3-030-04179-3_31"},{"key":"3163_CR15","unstructured":"Vidal R, Bruna J, Giryes R, et al. Mathematics of deep learning. 2017. ArXiv:1712.04741"},{"key":"3163_CR16","doi-asserted-by":"publisher","first-page":"30","DOI":"10.1007\/s40687-018-0148-y","volume":"5","author":"P Chaudhari","year":"2018","unstructured":"Chaudhari P, Oberman A, Osher S, et al. Deep relaxation: partial differential equations for optimizing deep neural networks. Res Math Sci, 2018, 5: 30","journal-title":"Res Math Sci"},{"key":"3163_CR17","doi-asserted-by":"publisher","first-page":"5079","DOI":"10.1109\/TNNLS.2019.2963066","volume":"31","author":"H Q Wang","year":"2020","unstructured":"Wang H Q, Luo Y, An W P, et al. PID controller-based stochastic optimization acceleration for deep neural networks. IEEE Trans Neural Netw Learn Syst, 2020, 31: 5079\u20135091","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"3163_CR18","doi-asserted-by":"publisher","first-page":"1313","DOI":"10.1109\/TNN.2008.2000391","volume":"19","author":"F Cousseau","year":"2008","unstructured":"Cousseau F, Ozeki T, Amari S. Dynamics of learning in multilayer perceptrons near singularities. IEEE Trans Neural Netw, 2008, 19: 1313\u20131328","journal-title":"IEEE Trans Neural Netw"},{"key":"3163_CR19","doi-asserted-by":"publisher","first-page":"1007","DOI":"10.1162\/neco.2006.18.5.1007","volume":"18","author":"S Amari","year":"2006","unstructured":"Amari S, Park H, Ozeki T. Singularities affect dynamics of learning in neuromanifolds. Neural Comput, 2006, 18: 1007\u20131065","journal-title":"Neural Comput"},{"key":"3163_CR20","first-page":"876","volume":"20","author":"A Bietti","year":"2019","unstructured":"Bietti A, Mairal J. Group invariance, stability to deformations, and complexity of deep convolutional representations. J Mach Learn Res, 2019, 20: 876\u2013924","journal-title":"J Mach Learn Res"},{"key":"3163_CR21","unstructured":"Sutskever I, Martens J, Dahl G, et al. On the importance of initialization and momentum in deep learning. In: Proceedings of International Conference on Machine Learning, 2013. 1139\u20131147"},{"key":"3163_CR22","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y Lecun","year":"1998","unstructured":"Lecun Y, Bottou L, Bengio Y, et al. Gradient-based learning applied to document recognition. Proc IEEE, 1998, 86: 2278\u20132324","journal-title":"Proc IEEE"},{"key":"3163_CR23","first-page":"1","volume":"18","author":"L S Li","year":"2018","unstructured":"Li L S, Jamieson K, DeSalvo G, et al. Hyperband: a novel bandit-based approach to hyperparameter optimization. J Mach Learn Res, 2018, 18: 1\u201352","journal-title":"J Mach Learn Res"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-020-3163-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-020-3163-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-020-3163-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,6,19]],"date-time":"2022-06-19T20:05:20Z","timestamp":1655669120000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-020-3163-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,4,8]]},"references-count":23,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2021,5]]}},"alternative-id":["3163"],"URL":"https:\/\/doi.org\/10.1007\/s11432-020-3163-0","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"value":"1674-733X","type":"print"},{"value":"1869-1919","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,4,8]]},"assertion":[{"value":"26 April 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 August 2020","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 November 2020","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 April 2021","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"150102"}}