{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T04:28:11Z","timestamp":1784867291935,"version":"3.55.0"},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2023,7,4]],"date-time":"2023-07-04T00:00:00Z","timestamp":1688428800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,7,4]],"date-time":"2023-07-04T00:00:00Z","timestamp":1688428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Numer Algor"],"published-print":{"date-parts":[[2024,1]]},"DOI":"10.1007\/s11075-023-01575-0","type":"journal-article","created":{"date-parts":[[2023,7,10]],"date-time":"2023-07-10T13:05:00Z","timestamp":1688994300000},"page":"383-421","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Theoretical analysis of Adam using hyperparameters close to one without Lipschitz smoothness"],"prefix":"10.1007","volume":"95","author":[{"given":"Hideaki","family":"Iiduka","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,7,4]]},"reference":[{"key":"1575_CR1","unstructured":"Arjovsky, M., Chintala, S., Bottou, L.: Wasserstein GAN https:\/\/arxiv.org\/pdf\/1701.07875.pdf (2017)"},{"key":"1575_CR2","doi-asserted-by":"crossref","unstructured":"Borwein, J.M., Lewis, A.S.: Convex Analysis and Nonlinear Optimization: Theory and Examples. Springer, New York (2000)","DOI":"10.1007\/978-1-4757-9859-3"},{"key":"1575_CR3","doi-asserted-by":"publisher","first-page":"223","DOI":"10.1137\/16M1080173","volume":"60","author":"L Bottou","year":"2018","unstructured":"Bottou, L., Curtis, F.E., Nocedal, J.: Optimization methods for large-scale machine learning. SIAM Review 60, 223\u2013311 (2018)","journal-title":"SIAM Review"},{"key":"1575_CR4","unstructured":"Chen, H., Zheng, L., AL\u00a0Kontar, R., Raskutti, G.: Stochastic gradient descent in correlated settings: A study on Gaussian processes. In: Advances in Neural Information Processing Systems, vol.\u00a033 (2020)"},{"key":"1575_CR5","doi-asserted-by":"crossref","unstructured":"Chen, J., Zhou, D., Tang, Y., Yang, Z., Cao, Y., Gu, Q.: Closing the generalization gap of adaptive gradient methods in training deep neural network. In: Proceedings of the Twenty-Ninth International Joint Conference on Artificial Intelligence, vol. 452, pp. 3267\u20133275 (2021)","DOI":"10.24963\/ijcai.2020\/452"},{"key":"1575_CR6","unstructured":"Chen, X., Liu, S., Sun, R., Hong, M.: On the convergence of a class of Adam-type algorithms for non-convex optimization. In: Proceedings of The International Conference on Learning Representations (2019)"},{"key":"1575_CR7","first-page":"2121","volume":"12","author":"J Duchi","year":"2011","unstructured":"Duchi, J., Hazan, E., Singer, Y.: Adaptive subgradient methods for online learning and stochastic optimization. Journal of Machine Learning Research 12, 2121\u20132159 (2011)","journal-title":"Journal of Machine Learning Research"},{"key":"1575_CR8","first-page":"1","volume":"21","author":"B Fehrman","year":"2020","unstructured":"Fehrman, B., Gess, B., Jentzen, A.: Convergence rates for the stochastic gradient descent method for non-convex objective functions. Journal of Machine Learning Research 21, 1\u201348 (2020)","journal-title":"Journal of Machine Learning Research"},{"key":"1575_CR9","doi-asserted-by":"crossref","unstructured":"Ghadimi, S., Lan, G.: Optimal stochastic approximation algorithms for strongly convex stochastic composite optimization I: A generic algorithmic framework. SIAM Journal on Optimization 22, 1469\u20131492 (2012)","DOI":"10.1137\/110848864"},{"key":"1575_CR10","doi-asserted-by":"publisher","first-page":"2061","DOI":"10.1137\/110848876","volume":"23","author":"S Ghadimi","year":"2013","unstructured":"Ghadimi, S., Lan, G.: Optimal stochastic approximation algorithms for strongly convex stochastic composite optimization II: Shrinking procedures and optimal algorithms. SIAM Journal on Optimization 23, 2061\u20132089 (2013)","journal-title":"SIAM Journal on Optimization"},{"key":"1575_CR11","unstructured":"Gower, R.M., Sebbouh, O., Loizou, N.: SGD for structured nonconvex functions: Learning rates, minibatching and interpolation. In: Proceedings of the 24th International Conference on Artificial Intelligence and Statistics, vol. 130 (2021)"},{"key":"1575_CR12","doi-asserted-by":"crossref","unstructured":"Horn, R.A., Johnson, C.R.: Matrix Analysis. Cambridge University Press, Cambridge (1985)","DOI":"10.1017\/CBO9780511810817"},{"issue":"12","key":"1575_CR13","doi-asserted-by":"publisher","first-page":"13250","DOI":"10.1109\/TCYB.2021.3107415","volume":"52","author":"H Iiduka","year":"2022","unstructured":"Iiduka, H.: Appropriate learning rates of adaptive learning rate optimization algorithms for training deep neural networks. IEEE Transactions on Cybernetics 52(12), 13250\u201313261 (2022)","journal-title":"IEEE Transactions on Cybernetics"},{"key":"1575_CR14","unstructured":"Kingma, D.P., Ba, J.: Adam: A method for stochastic optimization. In: Proceedings of The International Conference on Learning Representations (2015)"},{"key":"1575_CR15","unstructured":"Loizou, N., Vaswani, S., Laradji, I., Lacoste-Julien, S.: Stochastic polyak step-size for SGD: An adaptive learning rate for fast convergence. In: Proceedings of the 24th International Conference on Artificial Intelligence and Statistics, vol. 130 (2021)"},{"key":"1575_CR16","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: International Conference on Learning Representations (2019)"},{"key":"1575_CR17","unstructured":"Luo, L., Xiong, Y., Liu, Y., Sun, X.: Adaptive gradient methods with dynamic bound of learning rate. In: Proceedings of The International Conference on Learning Representations (2019)"},{"key":"1575_CR18","unstructured":"Mendler-D\u00fcnner, C., Perdomo, J.C., Zrnic, T., Hardt, M.: Stochastic optimization for performative prediction. In: Advances in Neural Information Processing Systems, vol.\u00a033 (2020)"},{"key":"1575_CR19","doi-asserted-by":"publisher","first-page":"1574","DOI":"10.1137\/070704277","volume":"19","author":"A Nemirovski","year":"2009","unstructured":"Nemirovski, A., Juditsky, A., Lan, G., Shapiro, A.: Robust stochastic approximation approach to stochastic programming. SIAM Journal on Optimization 19, 1574\u20131609 (2009)","journal-title":"SIAM Journal on Optimization"},{"key":"1575_CR20","first-page":"543","volume":"269","author":"Y Nesterov","year":"1983","unstructured":"Nesterov, Y.: A method for unconstrained convex minimization problem with the rate of convergence $${O}(1\/k^2)$$. Doklady AN USSR 269, 543\u2013547 (1983)","journal-title":"Doklady AN USSR"},{"key":"1575_CR21","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/0041-5553(64)90137-5","volume":"4","author":"BT Polyak","year":"1964","unstructured":"Polyak, B.T.: Some methods of speeding up the convergence of iteration methods. USSR Computational Mathematics and Mathematical Physics 4, 1\u201317 (1964)","journal-title":"USSR Computational Mathematics and Mathematical Physics"},{"key":"1575_CR22","unstructured":"Reddi, S.J., Kale, S., Kumar, S.: On the convergence of Adam and beyond. In: Proceedings of The International Conference on Learning Representations (2018)"},{"key":"1575_CR23","doi-asserted-by":"publisher","first-page":"400","DOI":"10.1214\/aoms\/1177729586","volume":"22","author":"H Robbins","year":"1951","unstructured":"Robbins, H., Monro, H.: A stochastic approximation method. The Annals of Mathematical Statistics 22, 400\u2013407 (1951)","journal-title":"The Annals of Mathematical Statistics"},{"key":"1575_CR24","unstructured":"Scaman, K., Malherbe, C.: Robustness analysis of non-convex stochastic gradient descent using biased expectations. In: Advances in Neural Information Processing Systems, vol.\u00a033 (2020)"},{"key":"1575_CR25","first-page":"1","volume":"20","author":"CJ Shallue","year":"2019","unstructured":"Shallue, C.J., Lee, J., Antognini, J., Sohl-Dickstein, J., Frostig, R., Dahl, G.E.: Measuring the effects of data parallelism on neural network training. Journal of Machine Learning Research 20, 1\u201349 (2019)","journal-title":"Journal of Machine Learning Research"},{"key":"1575_CR26","unstructured":"Smith, S.L., Kindermans, P.J., Le, Q.V.: Don\u2019t decay the learning rate, increase the batch size. In: Proceedings of The International Conference on Learning Representations (2018)"},{"key":"1575_CR27","unstructured":"Tieleman, T., Hinton, G.: RMSProp: Divide the gradient by a running average of its recent magnitude. COURSERA: Neural networks for machine learning 4, 26\u201331 (2012)"},{"key":"1575_CR28","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, L., Polosukhin, I.: Attention is All you Need. In: Advances in Neural Information Processing Systems, vol.\u00a030 (2017)"},{"key":"1575_CR29","unstructured":"Virmaux, A., Scaman, K.: Lipschitz regularity of deep neural networks: analysis and efficient estimation. In: Advances in Neural Information Processing Systems, vol.\u00a031 (2018)"},{"key":"1575_CR30","unstructured":"Xu, K., Ba, J., Kiros, R., Cho, K., Courville, A., Salakhudinov, R., Zemel, R., Bengio, Y.: Show, attend and tell: Neural image caption generation with visual attention. In: Proceedings of the 32nd International Conference on Machine Learning, vol.\u00a037, pp. 2048\u20132057 (2015)"},{"key":"1575_CR31","unstructured":"Zaheer, M., Reddi, S., Sachan, D., Kale, S., Kumar, S.: Adaptive methods for nonconvex optimization. In: S.\u00a0Bengio, H.\u00a0Wallach, H.\u00a0Larochelle, K.\u00a0Grauman, N.\u00a0Cesa-Bianchi, R.\u00a0Garnett (eds.) Advances in Neural Information Processing Systems, vol.\u00a031. Curran Associates, Inc. (2018)"},{"key":"1575_CR32","unstructured":"Zhang, G., Li, L., Nado, Z., Martens, J., Sachdeva, S., Dahl, G.E., Shallue, C.J., Grosse, R.: Which algorithmic choices matter at which batch sizes? Insights from a noisy quadratic model. In: Advances in Neural Information Processing Systems, vol.\u00a032 (2019)"},{"key":"1575_CR33","unstructured":"Zhou, D., Chen, J., Cao, Y., Tang, Y., Yang, Z., Gu, Q.: On the convergence of adaptive gradient methods for nonconvex optimization. In: 12th Annual Workshop on Optimization for Machine Learning (2020)"},{"key":"1575_CR34","unstructured":"Zhuang, J., Tang, T., Ding, Y., Tatikonda, S., Dvornek, N., Papademetris, X., Duncan, J.S.: AdaBelief optimizer: Adapting stepsizes by the belief in observed gradients. In: Advances in Neural Information Processing Systems, vol.\u00a033 (2020)"},{"key":"1575_CR35","unstructured":"Zinkevich, M.: Online convex programming and generalized infinitesimal gradient ascent. In: Proceedings of the 20th International Conference on Machine Learning, pp. 928\u2013936 (2003)"},{"key":"1575_CR36","unstructured":"Zinkevich, M., Weimer, M., Li, L., Smola, A.: Parallelized stochastic gradient descent. In: Advances in Neural Information Processing Systems, vol.\u00a023 (2010)"},{"key":"1575_CR37","doi-asserted-by":"crossref","unstructured":"Zou, F., Shen, L., Jie, Z., Zhang Weizhong, W.L.: A sufficient condition for convergences of Adam and RMSProp. In: Computer Vision and Pattern Recognition Conference, pp. 11127\u201311135 (2019)","DOI":"10.1109\/CVPR.2019.01138"}],"container-title":["Numerical Algorithms"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11075-023-01575-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11075-023-01575-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11075-023-01575-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,4]],"date-time":"2024-01-04T10:35:11Z","timestamp":1704364511000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11075-023-01575-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,7,4]]},"references-count":37,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,1]]}},"alternative-id":["1575"],"URL":"https:\/\/doi.org\/10.1007\/s11075-023-01575-0","relation":{},"ISSN":["1017-1398","1572-9265"],"issn-type":[{"value":"1017-1398","type":"print"},{"value":"1572-9265","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,7,4]]},"assertion":[{"value":"22 December 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 May 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 July 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not Applicable","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}},{"value":"Not Applicable","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}