{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T06:52:29Z","timestamp":1774421549593,"version":"3.50.1"},"reference-count":30,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2022,12,9]],"date-time":"2022-12-09T00:00:00Z","timestamp":1670544000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,12,9]],"date-time":"2022-12-09T00:00:00Z","timestamp":1670544000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62002102"],"award-info":[{"award-number":["62002102"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["72002133"],"award-info":[{"award-number":["72002133"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62172142"],"award-info":[{"award-number":["62172142"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2023,4]]},"DOI":"10.1007\/s00521-022-08082-8","type":"journal-article","created":{"date-parts":[[2022,12,9]],"date-time":"2022-12-09T07:03:19Z","timestamp":1670569399000},"page":"8051-8063","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["SAdaBoundNc: an adaptive subgradient online learning algorithm with logarithmic regret bounds"],"prefix":"10.1007","volume":"35","author":[{"given":"Lin","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7552-455X","authenticated-orcid":false,"given":"Xin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tao","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruijuan","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junlong","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingchuan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,12,9]]},"reference":[{"key":"8082_CR1","doi-asserted-by":"crossref","unstructured":"Bottou L (2010) Large-scale machine learning with stochastic gradient descent. In: Proceedings of COMPSTAT\u20192010. Physica-Verlag, Heidelberg, pp 177\u2013186","DOI":"10.1007\/978-3-7908-2604-3_16"},{"key":"8082_CR2","volume-title":"Adaptive filter theory","author":"S Haykin","year":"2014","unstructured":"Haykin S (2014) Adaptive filter theory, 5th edn. Pearson Education, London","edition":"5"},{"key":"8082_CR3","doi-asserted-by":"publisher","first-page":"400","DOI":"10.1214\/aoms\/1177729586","volume":"22","author":"H Robbins","year":"1951","unstructured":"Robbins H, Monro S (1951) A stochastic approximation method. Ann Math Stat 22:400\u2013407","journal-title":"Ann Math Stat"},{"key":"8082_CR4","doi-asserted-by":"publisher","first-page":"504","DOI":"10.1126\/science.1127647","volume":"313","author":"GE Hinton","year":"2006","unstructured":"Hinton GE, Salakhutdinov RR (2006) Reducing the dimensionality of data with neural networks. Science 313:504\u2013507","journal-title":"Science"},{"key":"8082_CR5","first-page":"2121","volume":"12","author":"J Duchi","year":"2011","unstructured":"Duchi J, Hazan E, Singer Y (2011) Adaptive subgradient methods for online learning and stochastic optimization. J Mach Learn Res 12:2121\u20132159","journal-title":"J Mach Learn Res"},{"key":"8082_CR6","unstructured":"Tieleman T, Hinton G (2012) Rmsprop: Divide the gradient by a running average of its recent magnitude. COURSERA: Neural Netw Mach Learn 4:26\u201331"},{"key":"8082_CR7","unstructured":"Kingma DP, Ba J (2015) Adam: a method for stochastic optimization. In: Proceedings of the 3rd international conference on learning representations, ICLR 2015, San Diego, CA, USA, May 7\u20139, Conference Track Proceedings, 2015. arxiv:1412.6980"},{"key":"8082_CR8","unstructured":"Wilson AC, Roelofs R, Stern M, Srebro N, Recht B (2017) The marginal value of adaptive gradient methods in machine learning. In: Proceedings of the 31st international conference on neural information processing systems, Curran Associates, Inc., pp 4148\u20134158"},{"key":"8082_CR9","unstructured":"Reddi SJ, Kale S, Kumar S (2018) On the convergence of Adam and beyond. In: Proceedings of the 6th international conference on learning representations, ICLR 2018, Vancouver, BC, Canada, April 30, May 3, Conference Track Proceedings. www.OpenReview.net, 2018. https:\/\/openreview.net\/forum?id=ryQu7f-RZ"},{"key":"8082_CR10","unstructured":"Luo L, Xiong Y, Liu Y, Sun X (2019) Adaptive gradient methods with dynamic bound of learning rate,. In: Proceedings of the 7th international conference on learning representations, ICLR 2019, New Orleans, LA, USA, May 6\u20139, 2019, www.OpenReview.net. https:\/\/openreview.net\/forum?id=Bkg3g2R9FX"},{"key":"8082_CR11","doi-asserted-by":"publisher","unstructured":"Chen J, Zhou D, Tang Y, Yang Z, Cao Y, Gu Q (2020) Closing the generalization gap of adaptive gradient methods in training deep neural networks. In: Proceedings of the twenty-ninth international joint conference on artificial intelligence, IJCAI , www.ijcai.org 2020, pp 3267\u20133275. https:\/\/doi.org\/10.24963\/ijcai.2020\/452","DOI":"10.24963\/ijcai.2020\/452"},{"key":"8082_CR12","doi-asserted-by":"publisher","first-page":"157","DOI":"10.1561\/2400000013","volume":"2","author":"E Hazan","year":"2016","unstructured":"Hazan E (2016) Introduction to online convex optimization. Found Trends Optim 2:157\u2013325","journal-title":"Found Trends Optim"},{"key":"8082_CR13","doi-asserted-by":"publisher","first-page":"169","DOI":"10.1007\/s10994-007-5016-8","volume":"69","author":"E Hazan","year":"2007","unstructured":"Hazan E, Agarwal A, Kale S (2007) Logarithmic regret algorithms for online convex optimization. Mach Learn 69:169\u2013192","journal-title":"Mach Learn"},{"key":"8082_CR14","unstructured":"Mukkamala MC, Hein M (2017) Variants of rmsprop and adagrad with logarithmic regret bounds. In: Proceedings of the 34th international conference on machine learning, ICML 2017, Sydney,NSW, Australia, 6\u201311 August , vol 70 of proceedings of machine learning research, PMLR, 2017, pp 2545\u20132553. http:\/\/proceedings.mlr.press\/v70\/mukkamala17a.html"},{"key":"8082_CR15","unstructured":"Wang G, Lu S, Cheng Q, Tu W, Zhang L (2020) Sadam: a variant of adam for strongly convex functions. In: Proceedings of the 8th international conference on learning representations, ICLR 2020, Addis Ababa, Ethiopia, April 26\u201330, www.OpenReview.net, 2020. https:\/\/openreview.net\/forum?id=rye5YaEtPr"},{"key":"8082_CR16","unstructured":"Zinkevich M (2003) Online convex programming and generalized infinitesimal gradient ascent. In: Proceedings of the Twentieth International conference on machine learning (ICML 2003), August 21\u201324, Washington, DC, USA, AAAI Press, 2003, pp 928\u2013936. http:\/\/www.aaai.org\/Library\/ICML\/2003\/icml03-120.php"},{"key":"8082_CR17","doi-asserted-by":"publisher","first-page":"2050","DOI":"10.1109\/TIT.2004.833339","volume":"50","author":"N Cesa-Bianchi","year":"2004","unstructured":"Cesa-Bianchi N, Conconi A, Gentile C (2004) On the generalization ability of on-line learning algorithms. IEEE Trans Inf Theory 50:2050\u20132057","journal-title":"IEEE Trans Inf Theory"},{"key":"8082_CR18","unstructured":"Zeiler MD (2012) Adadelta: an adaptive learning rate method. CoRR arxiv:1212.5701"},{"key":"8082_CR19","unstructured":"Dozat T (2016) Incorporating nesterov momentum into adam. In: Proceedings of the 4th international conference on learning representations, ICLR 2016, San Juan, Puerto Rico, May 2\u20134, Workshop Track Proceedings,2016. https:\/\/openreview.net\/forum?id=OM0jvwB8jIp57Z-JjtNEZ &noteId=OM0jvwB8jIp57ZJjtNEZ"},{"key":"8082_CR20","unstructured":"Gregor K, Danihelka I, Graves A, Rezende DJ, Wierstra D (2015) DRAW: a recurrent neural network for image generation. In: Proceedings of the 32nd international conference on machine learning, ICML 2015, Lille, France, 6\u201311 July , vol 37 of JMLR workshop and conference proceedings, www.JMLR.org, 2015, pp. 1462\u20131471. http:\/\/proceedings.mlr.press\/v37\/gregor15.html"},{"key":"8082_CR21","unstructured":"Xu K, Ba J, Kiros R, Cho K, Courville AC, Salakhutdinov R, Zemel RS, Bengio Y (2015) Show, attend and tell: Neural image caption generation with visual attention. In: Proceedings of the 32nd international conference on machine learning, ICML 2015, Lille,France, 6\u201311 July , vol 37 of JMLR Workshop and conference proceedings, www.JMLR.org, 2015, pp 2048\u20132057. http:\/\/proceedings.mlr.press\/v37\/xuc15.html"},{"key":"8082_CR22","unstructured":"Choi D, Shallue CJ, Nado Z, Lee J, Maddison CJ, Dahl GE (2019) On empirical comparisons of optimizers for deep learning. CoRR arxiv:1910.05446"},{"key":"8082_CR23","unstructured":"Keskar NS, Socher R (2017) Improving generalization performance by switching from Adam to SGD. CoRR arxiv:1712.07628"},{"key":"8082_CR24","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511804441","volume-title":"Convex optimization","author":"S Boyd","year":"2004","unstructured":"Boyd S, Vandenberghe L (2004) Convex optimization. Cambridge University Press, Cambridge"},{"key":"8082_CR25","unstructured":"Krizhevsky A (2009) Learning multiple layers of features from tiny images. Technical Report, Citeseer"},{"key":"8082_CR26","doi-asserted-by":"publisher","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of 2016 IEEE conference on computer vision and pattern recognition, CVPR 2016, Las Vegas, NV, USA, June 27\u201330, IEEE Computer Society, 2016, pp. 770\u2013778. https:\/\/doi.org\/10.1109\/CVPR.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"8082_CR27","unstructured":"Chen Z, Xu Y, Chen E, Yang T (2018) SADAGRAD: strongly adaptive stochastic gradient method. In: Dy JG, Krause A (Eds.), Proceedings of the 35th international conference on machine learning, ICML 2018, Stockholmsm\u00e4ssan, Stockholm, Sweden, July 10\u201315, vol 80 of proceedings of machine learning research, PMLR, 2018, pp 912\u2013920. http:\/\/proceedings.mlr.press\/v80\/chen18m.html"},{"key":"8082_CR28","unstructured":"McMahan HB, Streeter MJ (2010) Adaptive bound optimization for online convex optimization. In: Proceedings of The 23rd conference on learning theory (COLT 2010), Haifa, Israel, June 27\u201329, Omnipress, 2010, pp. 244\u2013256. http:\/\/colt2010.haifa.il.ibm.com\/papers\/COLT2010proceedings.pdfpage=252"},{"key":"8082_CR29","doi-asserted-by":"crossref","unstructured":"Marcus M, Santorini B, Marcinkiewicz MA (1993) Building a large annotated corpus of English: The Penn Treebank","DOI":"10.21236\/ADA273556"},{"key":"8082_CR30","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2009","unstructured":"Everingham M, Gool LV, Williams CKI, Winn JM, Zisserman A (2009) The Pascal Visual Object Classes (VOC) challenge. Int J Comput Vision 88:303\u2013338","journal-title":"Int J Comput Vision"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-022-08082-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-022-08082-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-022-08082-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,3,21]],"date-time":"2023-03-21T12:18:29Z","timestamp":1679401109000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-022-08082-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12,9]]},"references-count":30,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2023,4]]}},"alternative-id":["8082"],"URL":"https:\/\/doi.org\/10.1007\/s00521-022-08082-8","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,12,9]]},"assertion":[{"value":"28 January 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 November 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 December 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declaration"}},{"value":"The authors declare that there is no conflict of interests regarding the publication of this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}