{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,25]],"date-time":"2026-02-25T05:54:29Z","timestamp":1771998869696,"version":"3.50.1"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"35","license":[{"start":{"date-parts":[[2024,9,18]],"date-time":"2024-09-18T00:00:00Z","timestamp":1726617600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,9,18]],"date-time":"2024-09-18T00:00:00Z","timestamp":1726617600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"FONDECYT","award":["174-2020-FONDECYT"],"award-info":[{"award-number":["174-2020-FONDECYT"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1007\/s00521-024-10302-2","type":"journal-article","created":{"date-parts":[[2024,9,24]],"date-time":"2024-09-24T06:03:00Z","timestamp":1727157780000},"page":"22223-22243","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Fine-tuning adaptive stochastic optimizers: determining the optimal hyperparameter $$\\epsilon$$ via gradient magnitude histogram analysis"],"prefix":"10.1007","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9497-8901","authenticated-orcid":false,"given":"Gustavo","family":"Silva","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8501-0907","authenticated-orcid":false,"given":"Paul","family":"Rodriguez","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,9,18]]},"reference":[{"key":"10302_CR1","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"10302_CR2","doi-asserted-by":"crossref","unstructured":"Manning CD, Surdeanu M, Bauer J, Finkel JR, Bethard S, McClosky D (2014) The stanford CoreNLP natural language processing toolkit. In: Proceedings of 52nd annual meeting of the Association for Computational Linguistics: system demonstrations, pp 55\u201360","DOI":"10.3115\/v1\/P14-5010"},{"issue":"7540","key":"10302_CR3","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller M, Fidjeland AK, Ostrovski G et al (2015) Human-level control through deep reinforcement learning. Nature 518(7540):529\u2013533. https:\/\/doi.org\/10.1038\/nature14236","journal-title":"Nature"},{"key":"10302_CR4","doi-asserted-by":"crossref","unstructured":"Bottou L (2012) In: Montavon G, Orr GB, M\u00fcller K-R (eds) Stochastic Gradient Descent Tricks. Springer, Berlin, Heidelberg, pp 421\u2013436","DOI":"10.1007\/978-3-642-35289-8_25"},{"key":"10302_CR5","doi-asserted-by":"crossref","unstructured":"Bengio Y (2012) In: Montavon G, Orr GB, M\u00fcller K-R (eds) Practical Recommendations for Gradient-Based Training of Deep Architectures. Springer, Berlin, Heidelberg, pp 437\u2013478","DOI":"10.1007\/978-3-642-35289-8_26"},{"key":"10302_CR6","unstructured":"Wilson AC, Roelofs R, Stern M, Srebro N, Recht B (2017) The marginal value of adaptive gradient methods in machine learning. In: Advances in neural information processing systems, vol. 30, pp 4151\u20134161"},{"key":"10302_CR7","unstructured":"Choi D, Shallue CJ, Nado Z, Lee J, Maddison CJ, Dahl GE (2019) On empirical comparisons of optimizers for deep learning. arXiv preprint arXiv:1910.05446"},{"key":"10302_CR8","doi-asserted-by":"crossref","unstructured":"Szegedy C, Vanhoucke V, Ioffe S, Shlens J, Wojna Z (2016) Rethinking the inception architecture for computer vision. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 2818\u20132826","DOI":"10.1109\/CVPR.2016.308"},{"key":"10302_CR9","doi-asserted-by":"crossref","unstructured":"Tan M, Chen B, Pang R, Vasudevan V, Sandler M, Howard A, Le QV (2019) Mnasnet: platform-aware neural architecture search for mobile. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 2820\u20132828","DOI":"10.1109\/CVPR.2019.00293"},{"key":"10302_CR10","unstructured":"Tan M, Le Q (2019) Efficientnet: rethinking model scaling for convolutional neural networks. In: International conference on machine learning, pp 6105\u20136114"},{"key":"10302_CR11","unstructured":"Zaheer M, Reddi S, Sachan D, Kale S, Kumar S (2018) Adaptive methods for nonconvex optimization. In: Advances in neural information processing systems, vol. 31"},{"key":"10302_CR12","unstructured":"Liu L, Belkin M, Hsieh C-J (2020) On the variance of the adaptive learning rate and beyond. In: Proceedings of the 8th international conference on learning representations (ICLR)"},{"key":"10302_CR13","doi-asserted-by":"publisher","unstructured":"Robbins H, Monro S (1951) A stochastic approximation method. Ann Math Stat 22(3):400\u2013407. https:\/\/doi.org\/10.1214\/aoms\/1177729586","DOI":"10.1214\/aoms\/1177729586"},{"issue":"1","key":"10302_CR14","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1016\/S0893-6080(98)00116-6","volume":"12","author":"N Qian","year":"1999","unstructured":"Qian N (1999) On the momentum term in gradient descent learning algorithms. Neural Netw 12(1):145\u2013151. https:\/\/doi.org\/10.1016\/S0893-6080(98)00116-6","journal-title":"Neural Netw"},{"key":"10302_CR15","unstructured":"Nesterov YE (1983) A method of solving a convex programming problem with convergence rate o(k$$^{2}$$). In: Doklady Akademii Nauk, vol 269. Russian Academy of Sciences, pp 543\u2013547"},{"key":"10302_CR16","unstructured":"Lucas J, Sun S, Zemel R, Grosse R (2019) Aggregated momentum: stability through passive damping. In: International conference on learning representations"},{"key":"10302_CR17","unstructured":"Ma J, Yarats D (2019) Quasi-hyperbolic momentum and adam for deep learning. In: International conference on learning representations"},{"key":"10302_CR18","doi-asserted-by":"crossref","unstructured":"Rodriguez P (2023) Improving the stochastic gradient descent\u2019s test accuracy by manipulating the $$\\ell _\\infty$$ norm of its gradient approximation. In: ICASSP 2023-2023 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 1\u20135","DOI":"10.1109\/ICASSP49357.2023.10096624"},{"key":"10302_CR19","doi-asserted-by":"crossref","unstructured":"Rodriguez P (2023) Assessment of a two-step integration method as an optimizer for deep learning. In: 2023 31st European signal processing conference (EUSIPCO). IEEE, pp 1245\u20131249","DOI":"10.23919\/EUSIPCO58844.2023.10289761"},{"key":"10302_CR20","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, Weissenborn D, Zhai X, Unterthiner T, Dehghani M, Minderer M, Heigold G, Gelly S, Uszkoreit J, Houlsby N (2021) An image is worth 16x16 words: Transformers for image recognition at scale. In: International Conference on Learning Representations"},{"key":"10302_CR21","doi-asserted-by":"crossref","unstructured":"Shen W, Wang X, Wang Y, Bai X, Zhang Z (2015) Deepcontour: a deep convolutional feature learned by positive-sharing loss for contour detection. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3982\u20133991","DOI":"10.1109\/CVPR.2015.7299024"},{"key":"10302_CR22","unstructured":"Keskar NS, Mudigere D, Nocedal J, Smelyanskiy M, Tang PTP (2017) On large-batch training for deep learning: generalization gap and sharp minima. In: International conference on learning representations"},{"issue":"61","key":"10302_CR23","first-page":"2121","volume":"12","author":"J Duchi","year":"2011","unstructured":"Duchi J, Hazan E, Singer Y (2011) Adaptive subgradient methods for online learning and stochastic optimization. J Mach Learn Res 12(61):2121\u20132159","journal-title":"J Mach Learn Res"},{"key":"10302_CR24","unstructured":"Tieleman T, Hinton G (2012) Lecture 6.5-rmsprop, coursera: Neural networks for machine learning. Technical report, University of Toronto"},{"key":"10302_CR25","unstructured":"Kingma D, Ba J (2014) Adam: a method for stochastic optimization. In: International conference on learning representations"},{"key":"10302_CR26","unstructured":"Zhuang J, Tang T, Ding Y, Tatikonda SC, Dvornek N, Papademetris X, Duncan J (2020) Adabelief optimizer: adapting stepsizes by the belief in observed gradients. In: Advances in neural information processing systems, vol. 33, pp. 18795\u201318806"},{"key":"10302_CR27","doi-asserted-by":"crossref","unstructured":"Smith LN, Topin N (2019) Super-convergence: very fast training of neural networks using large learning rates. In: Artificial intelligence and machine learning for multi-domain operations applications, vol 11006. SPIE, pp 369\u2013386","DOI":"10.1117\/12.2520589"},{"key":"10302_CR28","unstructured":"Robbins H, Siegmund D (1971) A convergence theorem for non negative almost supermartingales and some applications. In: Optimizing methods in statistics"},{"key":"10302_CR29","doi-asserted-by":"crossref","unstructured":"Smith LN (2017) Cyclical learning rates for training neural networks. In: 2017 IEEE winter conference on applications of computer vision (WACV). IEEE, pp 464\u2013472","DOI":"10.1109\/WACV.2017.58"},{"key":"10302_CR30","doi-asserted-by":"crossref","unstructured":"Hessel M, Modayil J, Van\u00a0Hasselt H, Schaul T, Ostrovski G, Dabney W, Horgan D, Piot B, Azar M, Silver D (2018) Rainbow: combining improvements in deep reinforcement learning. In: Proceedings of the AAAI conference on artificial intelligence, vol 32","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"10302_CR31","unstructured":"Martens J, Grosse R (2015) Optimizing neural networks with kronecker-factored approximate curvature. In: International conference on machine learning, pp 2408\u20132417"},{"key":"10302_CR32","doi-asserted-by":"crossref","unstructured":"Savarese P, McAllester D, Babu S, Maire M (2021) Domain-independent dominance of adaptive methods. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 16286\u201316295","DOI":"10.1109\/CVPR46437.2021.01602"},{"key":"10302_CR33","doi-asserted-by":"publisher","first-page":"187","DOI":"10.1016\/j.trc.2015.03.014","volume":"54","author":"X Ma","year":"2015","unstructured":"Ma X, Tao Z, Wang Y, Yu H, Wang Y (2015) Long short-term memory neural network for traffic speed prediction using remote microwave sensor data. Transp Res Part C Emerg Technol 54:187\u2013197. https:\/\/doi.org\/10.1016\/j.trc.2015.03.014","journal-title":"Transp Res Part C Emerg Technol"},{"key":"10302_CR34","unstructured":"Bergstra J, Bengio Y (2012) Random search for hyper-parameter optimization. J Mach Learn Res 13(10):281\u2013305"},{"key":"10302_CR35","unstructured":"Wang Y, Kang Y, Qin C, Wang H, Xu Y, Zhang Y, Fu Y (2021) Rethinking adam: a twofold exponential moving average approach. arXiv preprint arXiv:2106.11514"},{"key":"10302_CR36","unstructured":"Simonyan K, Zisserman A (2015) Very deep convolutional networks for large-scale image recognition. In: 3rd international conference on learning representations, ICLR 2015. San Diego, CA, USA, Conference Track Proceedings"},{"key":"10302_CR37","doi-asserted-by":"crossref","unstructured":"Huang G, Liu Z, Van Der\u00a0Maaten L, Weinberger KQ (2017) Densely connected convolutional networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4700\u20134708","DOI":"10.1109\/CVPR.2017.243"},{"key":"10302_CR38","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) ImageNet classification with deep convolutional neural networks. In: Advances in neural information processing systems, vol. 25"},{"key":"10302_CR39","doi-asserted-by":"publisher","DOI":"10.1093\/oso\/9780198538493.001.0001","volume-title":"Neural networks for pattern recognition","author":"CM Bishop","year":"1995","unstructured":"Bishop CM (1995) Neural networks for pattern recognition. Oxford University Press Inc, USA"},{"key":"10302_CR40","unstructured":"Krizhevsky A (2009) Learning multiple layers of features from tiny images. Technical report, University of Toronto"},{"issue":"7","key":"10302_CR41","first-page":"3","volume":"7","author":"Y Le","year":"2015","unstructured":"Le Y, Yang X (2015) Tiny imagenet visual recognition challenge. CS 231N 7(7):3","journal-title":"CS 231N"},{"issue":"2","key":"10302_CR42","doi-asserted-by":"publisher","first-page":"3713","DOI":"10.35940\/ijrte.B3092.078219","volume":"8","author":"K Greeshma","year":"2019","unstructured":"Greeshma K, Sreekumar K (2019) Hyperparameter optimization and regularization on Fashion-MNIST classification. Int J Recent Technol Eng (IJRTE) 8(2):3713\u20133719. https:\/\/doi.org\/10.35940\/ijrte.B3092.078219","journal-title":"Int J Recent Technol Eng (IJRTE)"},{"issue":"2","key":"10302_CR43","first-page":"313","volume":"19","author":"M Marcus","year":"1993","unstructured":"Marcus M, Santorini B, Marcinkiewicz MA (1993) Building a large annotated corpus of English: The Penn Treebank. Comput Linguist 19(2):313\u2013330","journal-title":"Comput Linguist"},{"key":"10302_CR44","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez A.N, Kaiser L, Polosukhin I (2017) Attention is all you need. In: Advances in neural information processing systems, vol. 30"},{"issue":"11","key":"10302_CR45","doi-asserted-by":"publisher","first-page":"4500","DOI":"10.1109\/TNNLS.2019.2955777","volume":"31","author":"SR Dubey","year":"2019","unstructured":"Dubey SR, Chakraborty S, Roy SK, Mukherjee S, Singh SK, Chaudhuri BB (2019) diffGrad: an optimization method for convolutional neural networks. IEEE Trans Neural Netw Learn Syst 31(11):4500\u20134511. https:\/\/doi.org\/10.1109\/TNNLS.2019.2955777","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"10302_CR46","unstructured":"Reddi SJ, Kale S, Kumar S (2018) On the convergence of adam and beyond. In: International conference on learning representations"},{"key":"10302_CR47","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9781316402276","volume-title":"Machine learning refined: foundations, algorithms, and applications","author":"J Watt","year":"2016","unstructured":"Watt J, Borhani R, Katsaggelos AK (2016) Machine learning refined: foundations, algorithms, and applications, 1st edn. Cambridge University Press, USA","edition":"1"},{"key":"10302_CR48","unstructured":"Choromanska A, Henaff M, Mathieu M, Arous GB, LeCun Y (2015) The loss surfaces of multilayer networks. In: Artificial intelligence and statistics. PMLR, pp 192\u2013204"},{"key":"10302_CR49","unstructured":"Dauphin YN, Pascanu R, Gulcehre C, Cho K, Ganguli S, Bengio Y (2014) Identifying and attacking the saddle point problem in high-dimensional non-convex optimization. In: Advances in neural information processing systems, vol. 27, pp. 2933\u20132941"},{"issue":"3","key":"10302_CR50","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1145\/3446776","volume":"64","author":"C Zhang","year":"2021","unstructured":"Zhang C, Bengio S, Hardt M, Recht B, Vinyals O (2021) Understanding deep learning (still) requires rethinking generalization. Commun ACM 64(3):107\u2013115","journal-title":"Commun ACM"},{"key":"10302_CR51","unstructured":"Wang Y, Lacotte J, Pilanci M (2021) The hidden convex optimization landscape of regularized two-layer relu networks: an exact characterization of optimal solutions. In: International conference on learning representations"},{"key":"10302_CR52","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T, Zhu W-J (2002) Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th annual meeting of the association for computational linguistics, pp 311\u2013318","DOI":"10.3115\/1073083.1073135"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-024-10302-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-024-10302-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-024-10302-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,25]],"date-time":"2024-11-25T12:04:02Z","timestamp":1732536242000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-024-10302-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,9,18]]},"references-count":52,"journal-issue":{"issue":"35","published-print":{"date-parts":[[2024,12]]}},"alternative-id":["10302"],"URL":"https:\/\/doi.org\/10.1007\/s00521-024-10302-2","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,9,18]]},"assertion":[{"value":"4 March 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 July 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 September 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that there is no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}