{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T14:58:35Z","timestamp":1782399515642,"version":"3.54.5"},"reference-count":36,"publisher":"Springer Science and Business Media LLC","issue":"15","license":[{"start":{"date-parts":[[2023,5,9]],"date-time":"2023-05-09T00:00:00Z","timestamp":1683590400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,5,9]],"date-time":"2023-05-09T00:00:00Z","timestamp":1683590400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2023,10]]},"DOI":"10.1007\/s11227-023-05338-5","type":"journal-article","created":{"date-parts":[[2023,5,10]],"date-time":"2023-05-10T21:54:11Z","timestamp":1683755651000},"page":"17691-17715","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["AdaXod: a new adaptive and momental bound algorithm for training deep neural networks"],"prefix":"10.1007","volume":"79","author":[{"given":"Yuanxuan","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dequan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,5,9]]},"reference":[{"key":"5338_CR1","doi-asserted-by":"publisher","first-page":"377","DOI":"10.1016\/j.procs.2018.05.198","volume":"132","author":"Neha Sharma","year":"2018","unstructured":"Sharma Neha, Jain Vibhor, Mishra Anju (2018) An analysis of convolutional neural networks for image classification. Procedia Comput Sci 132:377\u2013384","journal-title":"Procedia Comput Sci"},{"key":"5338_CR2","first-page":"942","volume":"26","author":"Szegedy Christian","year":"2013","unstructured":"Christian Szegedy, Alexander Toshev, Dumitru Erhan (2013) Deep neural networks for object detection. Adv Neural Inform Proc Syst 26:942","journal-title":"Adv Neural Inform Proc Syst"},{"issue":"2","key":"5338_CR3","doi-asserted-by":"publisher","first-page":"206","DOI":"10.1109\/JSTSP.2019.2908700","volume":"13","author":"Hendrik Purwins","year":"2019","unstructured":"Purwins Hendrik, Li Bo, Virtanen Tuomas, Schl\u00fcter Jan, Chang Shuo-Yiin, Sainath Tara (2019) Deep learning for audio signal processing. IEEE J Select Topics Signal Proc 13(2):206\u2013219","journal-title":"IEEE J Select Topics Signal Proc"},{"issue":"1","key":"5338_CR4","doi-asserted-by":"publisher","first-page":"973","DOI":"10.1007\/s11227-020-03321-y","volume":"77","author":"Bur\u00e7ak Kadir Can","year":"2021","unstructured":"Can Bur\u00e7ak Kadir, Kaan Baykan \u00d6mer, Harun U\u011fuz (2021) A new deep convolutional neural network model for classifying breast cancer histopathological images and the hyperparameter optimisation of the proposed model. J Supercomput 77(1):973\u2013989","journal-title":"J Supercomput"},{"issue":"12","key":"5338_CR5","doi-asserted-by":"publisher","first-page":"13911","DOI":"10.1007\/s11227-021-03838-w","volume":"77","author":"Ishaani Priyadarshini","year":"2021","unstructured":"Priyadarshini Ishaani, Cotton Chase (2021) A novel lstm-cnn-grid search-based deep neural network for sentiment analysis. J Supercomput 77(12):13911\u201313932","journal-title":"J Supercomput"},{"issue":"10","key":"5338_CR6","doi-asserted-by":"publisher","first-page":"10773","DOI":"10.1007\/s11227-021-03690-y","volume":"77","author":"Luu-Ngoc Do","year":"2021","unstructured":"Do Luu-Ngoc, Yang Hyung-Jeong, Nguyen Hai-Duong, Kim Soo-Hyung, Lee Guee-Sang, Na In-Seop (2021) Deep neural network-based fusion model for emotion recognition using visual data. J Supercomput 77(10):10773\u201310790","journal-title":"J Supercomput"},{"key":"5338_CR7","unstructured":"McMahan H\u00a0Brendan, Streeter Matthew (2010) Adaptive bound optimization for online convex optimization. arXiv preprintarXiv:1002.4908"},{"key":"5338_CR8","unstructured":"Sutskever Ilya, Martens James, Dahl George, Hinton Geoffrey (2013) On the importance of initialization and momentum in deep learning. In International conference on machine learning pages 1139\u20131147. PMLR"},{"issue":"12","key":"5338_CR9","doi-asserted-by":"crossref","first-page":"3071","DOI":"10.1109\/TPAMI.2018.2868685","volume":"41","author":"Long Mingsheng","year":"2018","unstructured":"Mingsheng Long, Yue Cao, Zhangjie Cao, Jianmin Wang, Jordan Michael I (2018) Transferable representation learning with deep adaptation networks. IEEE Trans Pattern Anal Machine Intell 41(12):3071\u20133085","journal-title":"IEEE Trans Pattern Anal Machine Intell"},{"key":"5338_CR10","doi-asserted-by":"publisher","first-page":"778","DOI":"10.1007\/s12559-018-9566-9","volume":"11","author":"Yang Xi","year":"2019","unstructured":"Xi Yang, Kaizhu Huang, Rui Zhang, Goulermas John Y (2019) A novel deep density model for unsupervised learning. Cognitive Comput 11:778\u2013788","journal-title":"Cognitive Comput"},{"key":"5338_CR11","first-page":"1","volume":"730","author":"Gui Yangting","year":"2022","unstructured":"Yangting Gui, Dequan Li, Runyue Fang (2022) A fast adaptive algorithm for training deep neural networks. Appl Intell 730:1\u201310","journal-title":"Appl Intell"},{"key":"5338_CR12","doi-asserted-by":"crossref","unstructured":"Robbins Herbert, Monro Sutton (1951) A stochastic approximation method. The Annals Mathemat Stat pages 400\u2013407","DOI":"10.1214\/aoms\/1177729586"},{"key":"5338_CR13","unstructured":"Balcan Maria-Florina, Khodak Mikhail, Talwalkar Ameet (2019) Provable guarantees for gradient-based meta-learning. In : International Conference on Machine Learning pages 424\u2013433. PMLR"},{"key":"5338_CR14","first-page":"543","volume":"269","author":"Yurii Nesterov","year":"1983","unstructured":"Nesterov Yurii (1983) A method for unconstrained convex minimization problem with the rate of convergence o (1\/k$$\\hat{\\,}$$ 2). In Doklady an ussr 269:543\u2013547","journal-title":"In Doklady an ussr"},{"key":"5338_CR15","unstructured":"Tieleman Tijmen,\u00a0Hinton G (2017) Divide the gradient by a running average of its recent magnitude. coursera: neural networks for machine learning. Technical report"},{"key":"5338_CR16","unstructured":"Duchi John, Hazan Elad, Singer Yoram (2011) Adaptive subgradient methods for online learning and stochastic optimization. J Machine Learn Res 12(7)"},{"key":"5338_CR17","doi-asserted-by":"crossref","unstructured":"Ghadimi Euhanna, Feyzmahdavian Hamid\u00a0Reza, Johansson Mikael (2015) Global convergence of the heavy-ball method for convex optimization. In: 2015 European Control Conference (ECC), pages 310\u2013315. IEEE","DOI":"10.1109\/ECC.2015.7330562"},{"issue":"2","key":"5338_CR18","doi-asserted-by":"publisher","first-page":"237","DOI":"10.1016\/0893-6080(94)00067-V","volume":"8","author":"J Perantonis Stavros","year":"1995","unstructured":"Perantonis Stavros J, Karras Dimitris A (1995) An efficient constrained learning algorithm with momentum acceleration. Neural Networks 8(2):237\u2013249","journal-title":"Neural Networks"},{"issue":"5","key":"5338_CR19","first-page":"566","volume":"6","author":"Agnes Lydia","year":"2019","unstructured":"Lydia Agnes, Francis Sagayaraj (2019) Adagrad-an optimizer for stochastic gradient descent. Int J Inf Comput Sci 6(5):566\u2013568","journal-title":"Int J Inf Comput Sci"},{"key":"5338_CR20","doi-asserted-by":"crossref","unstructured":"Zou Fangyu,\u00a0Shen Li, Jie Zequn, Zhang Weizhong, Liu Wei (2019) A sufficient condition for convergences of adam and rmsprop. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 11127\u201311135","DOI":"10.1109\/CVPR.2019.01138"},{"key":"5338_CR21","unstructured":"Kingma Diederik\u00a0P, Ba Jimmy (2014) Adam: A method for stochastic optimization. arXiv preprintarXiv:1412.6980"},{"key":"5338_CR22","unstructured":"Zhou Zhiming, Zhang Qingru, Lu Guansong, Wang Hongwei, Zhang Weinan, Yu Yong (2018) Adashift: Decorrelation and convergence of adaptive learning rate methods. arXiv preprintarXiv:1810.00143"},{"key":"5338_CR23","unstructured":"Savarese Pedro (2019) On the convergence of adabound and its connection to sgd. arXiv preprintarXiv:1908.04457"},{"key":"5338_CR24","unstructured":"Li Wenjie, Zhang Zhaoyang, Wang Xinjiang, Luo Ping (2020) Adax: Adaptive gradient descent with exponential long term memory. arXiv preprintarXiv:2004.09740"},{"key":"5338_CR25","unstructured":"Reddi Sashank\u00a0J, Kale Satyen, Kumar Sanjiv (2019) On the convergence of adam and beyond. arXiv preprintarXiv:1904.09237"},{"key":"5338_CR26","doi-asserted-by":"crossref","unstructured":"Tran Phuong\u00a0Thi, et\u00a0al (2019) On the convergence proof of amsgrad and a new version. IEEE Access 7: 61706\u201361716","DOI":"10.1109\/ACCESS.2019.2916341"},{"key":"5338_CR27","first-page":"18795","volume":"33","author":"Zhuang Juntang","year":"2020","unstructured":"Juntang Zhuang, Tommy Tang, Yifan Ding, Tatikonda Sekhar C, Nicha Dvornek, Xenophon Papademetris, James Duncan (2020) Adabelief optimizer: adapting stepsizes by the belief in observed gradients. Adv Neural Inform Proc Syst 33:18795\u201318806","journal-title":"Adv Neural Inform Proc Syst"},{"key":"5338_CR28","unstructured":"Ding Jianbang, Ren Xuancheng, Luo Ruixuan,\u00a0Sun Xu (2019) An adaptive and momental bound method for stochastic learning. arXiv preprintarXiv:1910.12249"},{"key":"5338_CR29","doi-asserted-by":"crossref","unstructured":"Wang Fei, Jiang Mengqing, Qian Chen, Yang Shuo, Li Cheng, Zhang Honggang, Wang Xiaogang, Tang Xiaoou (2017) Residual attention network for image classification. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pages 3156\u20133164","DOI":"10.1109\/CVPR.2017.683"},{"key":"5338_CR30","doi-asserted-by":"crossref","unstructured":"Bansal Monika, Kumar Munish, Sachdeva Monika, Mittal Ajay (2021) Transfer learning for image classification using vgg19: Caltech-101 image data set. J Ambient Intell Humanized Comput pages 1\u201312","DOI":"10.1007\/s12652-021-03488-z"},{"key":"5338_CR31","unstructured":"Clanuwat Tarin, Bober-Irizar Mikel, Kitamoto Asanobu, Lamb Alex, Yamamoto Kazuaki, Ha David (2018) Deep learning for classical japanese literature. arXiv preprintarXiv:1812.01718"},{"key":"5338_CR32","unstructured":"Xiao Han, Rasul Kashif, Vollgraf Roland (2017) Fashion-mnist: a novel image dataset for benchmarking machine learning algorithms. arXiv preprintarXiv:1708.07747"},{"key":"5338_CR33","doi-asserted-by":"crossref","unstructured":"He Kaiming, Zhang Xiangyu, Ren Shaoqing, Sun Jian (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition pages 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"5338_CR34","doi-asserted-by":"crossref","unstructured":"Huang Gao, Liu Zhuang, Van Der\u00a0Maaten Laurens, Weinberger Kilian\u00a0Q (2017) Densely connected convolutional networks. In Proceedings of the IEEE conference on computer vision and pattern recognition pages 4700\u20134708","DOI":"10.1109\/CVPR.2017.243"},{"key":"5338_CR35","doi-asserted-by":"crossref","unstructured":"Khan Riaz\u00a0Ullah, Zhang Xiaosong, Kumar Rajesh, Aboagye Emelia\u00a0Opoku (2018) Evaluating the performance of resnet model based on image recognition. In: Proceedings of the 2018 International Conference on Computing and Artificial Intelligence pages 86\u201390","DOI":"10.1145\/3194452.3194461"},{"key":"5338_CR36","doi-asserted-by":"publisher","first-page":"4121","DOI":"10.1109\/JSTARS.2020.3009352","volume":"13","author":"Wei Tong","year":"2020","unstructured":"Tong Wei, Chen Weitao, Han Wei, Li Xianju, Wang Lizhe (2020) Channel-attention-based densenet network for remote sensing image scene classification. IEEE J Select Topics Appl Earth Observ Remote Sens 13:4121\u20134132","journal-title":"IEEE J Select Topics Appl Earth Observ Remote Sens"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-023-05338-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-023-05338-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-023-05338-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,30]],"date-time":"2023-08-30T15:22:57Z","timestamp":1693408977000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-023-05338-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,5,9]]},"references-count":36,"journal-issue":{"issue":"15","published-print":{"date-parts":[[2023,10]]}},"alternative-id":["5338"],"URL":"https:\/\/doi.org\/10.1007\/s11227-023-05338-5","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,5,9]]},"assertion":[{"value":"23 April 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 May 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"We declare that the authors have no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"This work does not involve in ethics, medicine, etc. We declare that this declaration is not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}}]}}