{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T17:33:57Z","timestamp":1782408837380,"version":"3.54.5"},"reference-count":28,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2025,5,28]],"date-time":"2025-05-28T00:00:00Z","timestamp":1748390400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,28]],"date-time":"2025-05-28T00:00:00Z","timestamp":1748390400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key R&D Program of China","doi-asserted-by":"crossref","award":["2018AAA0100300"],"award-info":[{"award-number":["2018AAA0100300"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Sichuan Science and Technology Program","award":["No. 2024YFHZ0026"],"award-info":[{"award-number":["No. 2024YFHZ0026"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2025,7]]},"DOI":"10.1007\/s10994-025-06797-y","type":"journal-article","created":{"date-parts":[[2025,5,28]],"date-time":"2025-05-28T17:40:30Z","timestamp":1748454030000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["A new adaptive gradient method with gradient decomposition"],"prefix":"10.1007","volume":"114","author":[{"given":"Zhou","family":"Shao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hang","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0000-834X","authenticated-orcid":false,"given":"Tong","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,5,28]]},"reference":[{"key":"6797_CR1","unstructured":"Alacaoglu, A., Malitsky, Y., Mertikopoulos, P., & Cevher, V. (2020). A new regret analysis for Adam-type algorithms. In International conference on machine learning (pp. 202\u2013210)."},{"key":"6797_CR2","unstructured":"Chen, X., Liu, S., Sun, R., & Hong, M. (2019). On the convergence of a class of Adam-type algorithms for non-convex optimization. In International conference on learning representations."},{"key":"6797_CR3","doi-asserted-by":"crossref","unstructured":"Chen, J., Zhou, D., Tang, Y., Yang, Z., Cao, Y., & Gu, Q. (2020). Closing the generalization gap of adaptive gradient methods in training deep neural networks. In International joint conferences on artificial intelligence.","DOI":"10.24963\/ijcai.2020\/452"},{"key":"6797_CR4","unstructured":"Duchi, J., Hazan, E., & Singer, Y. (2011). Adaptive subgradient methods for online learning and stochastic optimization. Journal of Machine Learning Research (JMLR)."},{"key":"6797_CR5","doi-asserted-by":"crossref","unstructured":"Graves, A., Mohamed, A.-r., & Hinton, G.\u00a0E. (2013). Speech recognition with deep recurrent neural networks. In International conference on acoustics, speech and signal processing (ICASSP) (pp. 6645\u20136649).","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"6797_CR6","doi-asserted-by":"publisher","first-page":"157","DOI":"10.1561\/2400000013","volume":"2","author":"E Hazan","year":"2016","unstructured":"Hazan, E. (2016). Introduction to online convex optimization. Foundations and Trends in Optimization, 2, 157\u2013325.","journal-title":"Foundations and Trends in Optimization"},{"key":"6797_CR7","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2016.90"},{"key":"6797_CR8","first-page":"199","volume-title":"On-line learning processes in artificial neural networks, Vol.\u00a051 of North-Holland Mathematical Library","author":"TM Heskes","year":"1993","unstructured":"Heskes, T. M., & Kappen, B. (1993). On-line learning processes in artificial neural networks, Vol.\u00a051 of North-Holland Mathematical Library (pp. 199\u2013233). Elsevier."},{"key":"6797_CR9","doi-asserted-by":"crossref","unstructured":"Huang, G., Liu, Z., van der Maaten, L., & Weinberger, K. Q. (2017). Densely connected convolutional networks. In Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2017.243"},{"key":"6797_CR10","unstructured":"Kingma, D.\u00a0P., & Ba, J.\u00a0L. (2015). Adam: A method for stochastic optimization. In Proceedings of the 3rd international conference on learning representations (ICLR)."},{"key":"6797_CR11","unstructured":"Krizhevsky, A., & Hinton, G. E. (2009). Learning multiple layers of features from tiny images. Technical report."},{"key":"6797_CR12","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G. E. (2012). ImageNet classification with deep convolutional neural networks. In Advances in neural information processing systems (NIPS) (pp. 1097\u20131105)."},{"key":"6797_CR13","doi-asserted-by":"crossref","unstructured":"Lecun, Y., Bottou, L., Bengio, Y., & Haffner, P. (1998). Gradient-based learning applied to document recognition. In Proceedings of the IEEE (pp. 2278\u20132324).","DOI":"10.1109\/5.726791"},{"key":"6797_CR14","doi-asserted-by":"crossref","unstructured":"LeCun, Y., Bottou, L., Orr, G., & M\u00fcller, K. (2012). Efficient backprop, lecture notes in computer science (pp. 9\u201348).","DOI":"10.1007\/978-3-642-35289-8_3"},{"key":"6797_CR15","unstructured":"Liu, H., & Tian, X. (2020). AEGD: Adaptive gradient descent with energy. In Control and optimization: Numerical algebra."},{"key":"6797_CR16","unstructured":"Liu, L., Jiang, H., He, P., Chen, W., Liu, X., Gao, J., & Han, J. (2020). On the variance of the adaptive learning rate and beyond. In Proceedings of the eighth international conference on learning representations (ICLR 2020)."},{"key":"6797_CR17","unstructured":"Loshchilov, I., & Hutter, F. (2017). Decoupled weight decay regularization. In International conference on learning representations."},{"key":"6797_CR18","unstructured":"Luo, L., Xiong, Y., Liu, Y., & Sun, X. (2019). Adaptive gradient methods with dynamic bound of learning rate. In Proceedings of international conference on learning representations (ICLR)."},{"key":"6797_CR19","unstructured":"Malitsky, Y., & Mishchenko, K. (2019). Adaptive gradient descent without descent. In International conference on machine learning."},{"key":"6797_CR20","first-page":"313","volume":"19","author":"MP Marcus","year":"1993","unstructured":"Marcus, M. P., Santorini, B., & Marcinkiewicz, M. A. (1993). Building a large annotated corpus of English: The Penn Treebank. Computational Linguistics, 19, 313\u2013330.","journal-title":"Computational Linguistics"},{"key":"6797_CR21","first-page":"372","volume":"27","author":"Y Nesterov","year":"1983","unstructured":"Nesterov, Y. (1983). A method of solving a convex programming problem with convergence rate O(1\/sqrt(k)). Soviet Mathematics Doklady, 27, 372\u2013376.","journal-title":"Soviet Mathematics Doklady"},{"key":"6797_CR22","doi-asserted-by":"publisher","first-page":"791","DOI":"10.1016\/0041-5553(64)90137-5","volume":"4","author":"BT Polyak","year":"1964","unstructured":"Polyak, B. T. (1964). Some methods of speeding up the convergence of iteration methods. USSR Computational Mathematics and Mathematical Physics, 4, 791\u2013803.","journal-title":"USSR Computational Mathematics and Mathematical Physics"},{"key":"6797_CR23","unstructured":"Reddi, S.\u00a0J., Kale, S., & Kumar, S. (2018). On the convergence of Adam and beyond. In Proceedings of international conference on learning representations (ICLR)."},{"key":"6797_CR24","unstructured":"Tieleman, T., & Hinton, G. (2012). RMSprop: Divide the gradient by a running average of its recent magnitude. In COURSERA: Neural networks for machine learning."},{"key":"6797_CR25","unstructured":"Zaheer, M., Reddi, S., Sachan, D., Kale, S., & Kumar, S. (2018). Adaptive methods for nonconvex optimization. In Advances in neural information processing systems (NIPS)."},{"key":"6797_CR26","unstructured":"Zeiler, M.\u00a0D. (2012). AdaDelta: An adaptive learning rate method. arXiv:1212.5701."},{"issue":"3","key":"6797_CR27","doi-asserted-by":"publisher","first-page":"279","DOI":"10.1002\/nme.5372","volume":"110","author":"J Zhao","year":"2017","unstructured":"Zhao, J., Wang, Q., & Yang, X. (2017). Numerical approximations for a phase field dendritic crystal growth model based on the invariant energy quadratizationapproach. International Journal for Numerical Methods in Engineering, 110(3), 279\u2013300.","journal-title":"International Journal for Numerical Methods in Engineering"},{"key":"6797_CR28","unstructured":"Zhuang, J., Tang, T., Ding, Y., Tatikonda, S., Dvornek, N., Papademetris, X., & Duncan, J. (2020). AdaBelief optimizer: Adapting stepsizes by the belief in observed gradients. Advances in Neural Information Processing Systems (NIPS)."}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-025-06797-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10994-025-06797-y","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-025-06797-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T00:03:00Z","timestamp":1779926580000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10994-025-06797-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,28]]},"references-count":28,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2025,7]]}},"alternative-id":["6797"],"URL":"https:\/\/doi.org\/10.1007\/s10994-025-06797-y","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"value":"0885-6125","type":"print"},{"value":"1573-0565","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5,28]]},"assertion":[{"value":"12 June 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 October 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 May 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 May 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"The authors declare to give consent for publication.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}],"article-number":"155"}}