{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:47:11Z","timestamp":1785649631026,"version":"3.56.0"},"publisher-location":"Cham","reference-count":37,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032316653","type":"print"},{"value":"9783032316660","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-3-032-31666-0_8","type":"book-chapter","created":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:45:08Z","timestamp":1785649508000},"page":"112-127","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Eva Optimizer: Escaping Low-Curvature Traps in\u00a0Deep Learning"],"prefix":"10.1007","author":[{"given":"Antonio","family":"Di Cecco","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Carlo","family":"Metta","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Andrea","family":"Papini","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Marco","family":"Fantozzi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Silvia Giulia","family":"Galfr\u00e9","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Michelangelo","family":"Vegli\u00f3","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Luigi Amedeo","family":"Bianchi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Maurizio","family":"Parton","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Francesco","family":"Morandin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,3]]},"reference":[{"key":"8_CR1","unstructured":"Bernstein, J., Mingard, C., Huang, K., Azizan, N., Yue, Y.: Automatic Gradient Descent: Deep Learning without Hyperparameters, (2023). arXiv:2304.05187"},{"key":"8_CR2","unstructured":"Bernstein, J., Wang, Y.X., Azizzadenesheli, K., Anandkumar, A.: signsgd: Compressed optimisation for non-convex problems. In: ICML. pp. 560\u2013569 (2018)"},{"key":"8_CR3","unstructured":"Bernstein, J., Zhao, J., Azizzadenesheli, K., Anandkumar, A.: signsgd with majority vote is communication efficient and fault tolerant, p. ICLR. (2019)"},{"key":"8_CR4","unstructured":"Brock, A., De, S., Smith, S.L., Simonyan, K.: High-performance large-scale image recognition without normalization. In: ICML 2021, vol. 139, pp. 1059\u20131071. ()"},{"key":"8_CR5","doi-asserted-by":"crossref","unstructured":"Chen, X., Liang, C., Huang, D., Real, E., Wang, K., Liu, Y., Pham, H., Dong, X., Luong, T., Hsieh, C.J., Lu, Y., Le, Q.V.: Symbolic discovery of optimization algorithms pp, pp. 49205\u201349233","DOI":"10.52202\/075280-2140"},{"key":"8_CR6","unstructured":"Chen, X., Liu, S., Sun, R., Hong, M.: On the convergence of a class of adam-type algorithms for non-convex optimization. ICLR, (2020)"},{"key":"8_CR7","unstructured":"Choi, D., Shallue, C.J., Nado, Z., Lee, J., Maddison, C.J., Dahl, G.E.: On empirical comparisons of optimizers for deep learning, (2019). arXiv:1910.05446"},{"key":"8_CR8","doi-asserted-by":"crossref","unstructured":"Di Cecco, A., Metta, C., Fantozzi, M., Morandin, F., Parton, M.: Glonets: Globally connected neural networks. Adv. in Intell. Data Anal 23, 53\u201364 (2024)","DOI":"10.1007\/978-3-031-58547-0_5"},{"key":"8_CR9","doi-asserted-by":"crossref","unstructured":"Di Cecco, A., Papini, A., Metta, C., Fantozzi, M., Galfr\u00e8, S.G., Morandin, F., Parton, M.: Switchpath: Enhancing exploration in neural networks learning dynamics. In: Discovery Science 2024, vol. 15243, Springer (2024)","DOI":"10.1007\/978-3-031-78977-9_18"},{"key":"8_CR10","unstructured":"Duchi, J., Hazan, E., Singer, Y.: Adaptive subgradient methods for online learning and stochastic optimization. J. of Machine Learn. Res 12, 2121\u20132159 (2011)"},{"key":"8_CR11","doi-asserted-by":"publisher","first-page":"190","DOI":"10.1287\/ijoc.1.3.190","volume":"1","author":"FW Glover","year":"1989","unstructured":"Glover, F.W.: Tabu search - part i. INFORMS J. Comput. 1, 190\u2013206 (1989)","journal-title":"INFORMS J. Comput."},{"key":"8_CR12","doi-asserted-by":"crossref","unstructured":"Grinsztajn, L., Oyallon, E., Varoquaux, G.: Why do tree-based models still outperform deep learning on tabular data? In: NeurIPS 2022, vol. 35, pp. 507\u2013520. ()","DOI":"10.52202\/068431-0037"},{"key":"8_CR13","unstructured":"Hao, S., Sukhbaatar, S., Su, D., Li, X., Hu, Z., Weston, J., Tian, Y.: Training large language models to reason in a continuous latent space, (2024). arXiv:2412.06769"},{"issue":"4","key":"8_CR14","doi-asserted-by":"publisher","first-page":"295","DOI":"10.1016\/0893-6080(88)90003-2","volume":"1","author":"RA Jacobs","year":"1988","unstructured":"Jacobs, R.A.: Increased rates of convergence through learning rate adaptation. Neural Netw. 1(4), 295\u2013307 (1988)","journal-title":"Neural Netw."},{"key":"8_CR15","unstructured":"Keskar, N.S., Socher, R.: Improving generalization performance by switching from adam to sgd, p. ICLR. (2017)"},{"key":"8_CR16","unstructured":"Kingma, D.P., Ba, J.: Adam: A method for stochastic optimization. In: ICLR 2015"},{"key":"8_CR17","unstructured":"Kunstner, F., Balles, L., Hennig, P.: Dissecting adam: The sign, magnitude and variance of stochastic gradients. In: ICML. pp. 5938\u20135949 (2021)"},{"key":"8_CR18","unstructured":"Lewkowycz, A., Bahri, Y., Dyer, E., Sohl-Dickstein, J., Gur-Ari, G.: The large learning rate phase of deep learning: the catapult mechanism. (). arXiv:2003.02218"},{"key":"8_CR19","unstructured":"Liu, L., Jiang, H., He, P., Chen, W., Liu, X., Gao, J., Han, J.: On the variance of the adaptive learning rate and beyond. In: ICLR 2020, Addis Abeba (2020)"},{"key":"8_CR20","doi-asserted-by":"crossref","unstructured":"Liu, Z., Mao, H., Wu, C.Y., Feichtenhofer, C., Darrell, T., Xie, S.: A ConvNet for the 2020s. IEEE\/CVF CVPR pp. 11976\u201311986 (2022)","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"8_CR21","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: ICLR 2019. ()"},{"issue":"5","key":"8_CR22","doi-asserted-by":"publisher","first-page":"1671","DOI":"10.4208\/cicp.OA-2020-0165","volume":"28","author":"L Lu","year":"2020","unstructured":"Lu, L., Yeonjong, S., Yanhui, S., George, K.: Dying relu and initialization: Theory and numerical examples. Comm. in Comput. Physics 28(5), 1671\u20131706 (2020)","journal-title":"Comm. in Comput. Physics"},{"key":"8_CR23","unstructured":"Luo, L., Xiong, Y., Liu, Y., Sun, X.: Adaptive gradient methods with dynamic bound of learning rate, p. ICLR. (2019)"},{"key":"8_CR24","unstructured":"Malladi, S., Wettig, A., Yu, D., Chen, D., Arora, S.: A kernel-based view of language model fine-tuning. ICML 2023, 23610\u201323641 (2023)"},{"key":"8_CR25","unstructured":"Pascanu, R., Mikolov, T., Bengio, Y.: On the difficulty of training recurrent neural networks. In: ICML 2013, (2013)"},{"key":"8_CR26","unstructured":"Reddi, S.J., Kale, S., Kumar, S.: On the Convergence of Adam and Beyond. In: ICLR 2018, (2018)"},{"key":"8_CR27","doi-asserted-by":"crossref","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., Chen, L.C.: MobileNetV2: Inverted Residuals and Linear Bottlenecks. In: IEEE CVPR (2018)","DOI":"10.1109\/CVPR.2018.00474"},{"key":"8_CR28","unstructured":"Schmidt, R.M., Schneider, F., Hennig, P.: Descending through a crowded valley-benchmarking deep learning optimizers. ICML (2021)"},{"key":"8_CR29","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., Wojna, Z.: Rethinking the Inception Architecture for Computer Vision. In: IEEE CVPR, pp. 2818\u20132826. (2016)","DOI":"10.1109\/CVPR.2016.308"},{"key":"8_CR30","unstructured":"Tan, M., Le, Q.V.: Rethinking Model Scaling for Convolutional Neural Networks. In: International Conference on Machine Learning (ICML), pp. 6105\u20136114. (2019)"},{"key":"8_CR31","unstructured":"Wilson, A.C., Roelofs, R., Stern, M., Srebro, N., Recht, B.: The marginal value of adaptive gradient methods in machine learning. In: NeurIPS 2017. pp. 4148\u20134158"},{"key":"8_CR32","doi-asserted-by":"publisher","first-page":"17","DOI":"10.1016\/j.neunet.2021.02.011","volume":"139","author":"D Xu","year":"2021","unstructured":"Xu, D., Zhang, S., Zhang, H.: Convergence of the RMSProp deep learning method with penalty for nonconvex optimization. Neural Net. 139, 17\u201323 (2021)","journal-title":"Neural Net."},{"issue":"1","key":"8_CR33","doi-asserted-by":"crossref","first-page":"297","DOI":"10.1038\/s41597-021-01071-x","volume":"8","author":"J Yang","year":"2021","unstructured":"Yang, J., Shi, R., Wei, D., Liu, Z., Zhao, L., Ke, B., Pfister, H., Ni, B.: Medmnist v2: A large-scale lightweight benchmark for 2d and 3d biomedical image classification. Scientific Data 8(1), 297 (2021)","journal-title":"Scientific Data"},{"key":"8_CR34","unstructured":"You, Y., Gitman, I., Ginsburg, B.: Large batch training of convolutional networks, (2017). arXiv:1708.03888"},{"key":"8_CR35","unstructured":"You, Y., Li, J., Reddi, S.J., Hseu, J., Kumar, S., Bhojanapalli, S., Song, X., Demmel, J., Keutzer, K., Hsieh, C.: Large batch optimization for deep learning: Training BERT in 76 minutes. ICLR 2020, Addis Abeba"},{"key":"8_CR36","unstructured":"Zaheer, M., Reddi, S.J., Sachan, D.S., Kale, S., Kumar, S.: Adaptive methods for nonconvex optimization. In: NeurIPS 2018. pp. 9815\u20139825 (2018)"},{"key":"8_CR37","unstructured":"Zhuang, J., Tang, T., Ding, Y., Tatikonda, S., et\u00a0al.: Adabelief optimizer: Adapting stepsizes by the belief in observed gradients. In: NeurIPS 2020 (2020)"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-31666-0_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:45:12Z","timestamp":1785649512000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-31666-0_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,3]]},"ISBN":["9783032316653","9783032316660"],"references-count":37,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-31666-0_8","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,3]]},"assertion":[{"value":"3 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lyon","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 August 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 August 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2026.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}