{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T10:26:47Z","timestamp":1774434407307,"version":"3.50.1"},"reference-count":38,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,1,31]],"date-time":"2026-01-31T00:00:00Z","timestamp":1769817600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,2,5]],"date-time":"2026-02-05T00:00:00Z","timestamp":1770249600000},"content-version":"vor","delay-in-days":5,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Complex Intell. Syst."],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1007\/s40747-026-02226-2","type":"journal-article","created":{"date-parts":[[2026,1,31]],"date-time":"2026-01-31T12:04:09Z","timestamp":1769861049000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A general learning rate improvement strategy for deep neural networks training"],"prefix":"10.1007","volume":"12","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-6982-6994","authenticated-orcid":false,"given":"Tingting","family":"Wu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qingwei","family":"Dong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zheng","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Manxue","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,31]]},"reference":[{"issue":"9","key":"2226_CR1","first-page":"142","volume":"17","author":"L Bottou","year":"1998","unstructured":"Bottou L (1998) Online learning and stochastic approximations. Online learning in neural networks 17(9):142","journal-title":"Online learning in neural networks"},{"key":"2226_CR2","doi-asserted-by":"crossref","unstructured":"Wu Y, Liu L, Bae J, Chow K-H, Iyengar A, Pu C, Wei W, Yu L, Zhang Q (2019) Demystifying learning rate policies for high accuracy training of deep neural networks. In: 2019 IEEE International Conference on Big Data (Big Data), pp. 1971\u20131980 . IEEE","DOI":"10.1109\/BigData47090.2019.9006104"},{"key":"2226_CR3","unstructured":"Loshchilov I, Hutter F (2016) Sgdr: Stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983"},{"key":"2226_CR4","unstructured":"Kingma DP, Ba J (2014) Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980"},{"key":"2226_CR5","unstructured":"Reddi SJ, Kale S, Kumar S (2019) On the convergence of adam and beyond. arXiv preprint arXiv:1904.09237"},{"key":"2226_CR6","unstructured":"Jacot A, Gabriel F, Hongler C (2018) Neural tangent kernel: Convergence and generalization in neural networks. Advances in neural information processing systems 31"},{"key":"2226_CR7","unstructured":"Cohen J, Kaur S, Li Y, Kolter JZ, Talwalkar A (2020) Gradient descent on neural networks typically occurs at the edge of stability. In: International Conference on Learning Representations"},{"key":"2226_CR8","unstructured":"Goyal P, Doll\u00e1r P, Girshick R, Noordhuis P, Wesolowski L, Kyrola A, Tulloch A, Jia Y, He K (2017) Accurate, large minibatch sgd: Training imagenet in 1 hour. arXiv preprint arXiv:1706.02677"},{"key":"2226_CR9","unstructured":"Gotmare A, Keskar NS, Xiong C, Socher R (2018) A closer look at deep learning heuristics: Learning rate restarts, warmup and distillation. In: International Conference on Learning Representations"},{"key":"2226_CR10","unstructured":"Mohtashami A, Jaggi M, Stich SU (2023) Special properties of gradient descent with large learning rates. In: International Conference on Machine Learning, pp. 25082\u201325104. PMLR"},{"key":"2226_CR11","unstructured":"You K, Long M, Wang J, Jordan MI (2019) How does learning rate decay help modern neural networks? arXiv preprint arXiv:1908.01878"},{"key":"2226_CR12","doi-asserted-by":"crossref","unstructured":"Bengio Y (2012) Practical recommendations for gradient-based training of deep architectures. In: Neural Networks: Tricks of the Trade: Second Edition, pp. 437\u2013478. Springer, ???","DOI":"10.1007\/978-3-642-35289-8_26"},{"issue":"12","key":"2226_CR13","doi-asserted-by":"publisher","first-page":"13250","DOI":"10.1109\/TCYB.2021.3107415","volume":"52","author":"H Iiduka","year":"2021","unstructured":"Iiduka H (2021) Appropriate learning rates of adaptive learning rate optimization algorithms for training deep neural networks. IEEE Transactions on Cybernetics 52(12):13250\u201313261","journal-title":"IEEE Transactions on Cybernetics"},{"key":"2226_CR14","doi-asserted-by":"crossref","unstructured":"Smith LN (2017) Cyclical learning rates for training neural networks. In: 2017 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 464\u2013472. IEEE","DOI":"10.1109\/WACV.2017.58"},{"key":"2226_CR15","doi-asserted-by":"crossref","unstructured":"Smith LN, Topin N (2019) Super-convergence: Very fast training of neural networks using large learning rates. In: Artificial Intelligence and Machine Learning for Multi-domain Operations Applications, vol. 11006, pp. 369\u2013386. SPIE","DOI":"10.1117\/12.2520589"},{"key":"2226_CR16","unstructured":"Wang Z, Zhang J (2022) Incremental pid controller-based learning rate scheduler for stochastic gradient descent. IEEE Transactions on Neural Networks and Learning Systems"},{"key":"2226_CR17","unstructured":"Duchi J, Hazan E, Singer Y (2011) Adaptive subgradient methods for online learning and stochastic optimization. Journal of machine learning research 12(7)"},{"key":"2226_CR18","unstructured":"Hinton G, Srivastava N, Swersky K, Tieleman T, Mohamed A (2012) Coursera: Neural networks for machine learning. Lecture 9c: Using noise as a regularizer"},{"key":"2226_CR19","unstructured":"Loshchilov I, Hutter F (2018) Fixing weight decay regularization in adam"},{"key":"2226_CR20","unstructured":"Liu L, Jiang H, He P, Chen W, Liu X, Gao J, Han J (2019) On the variance of the adaptive learning rate and beyond. In: International Conference on Learning Representations"},{"key":"2226_CR21","doi-asserted-by":"crossref","unstructured":"Yang E, Pan J, Wang X, Yu H, Shen L, Chen X, Xiao L, Jiang J, Guo G (2023) Adatask: A task-aware adaptive learning rate approach to multi-task learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 10745\u201310753","DOI":"10.1609\/aaai.v37i9.26275"},{"key":"2226_CR22","doi-asserted-by":"crossref","unstructured":"Kabiri H, Ghanou Y, Khalifi H, Casalino G (2024) Amadam: adaptive modifier of adam method. Knowledge and Information Systems, 1\u201332","DOI":"10.1007\/s10115-023-02052-9"},{"key":"2226_CR23","doi-asserted-by":"crossref","unstructured":"Daniel C, Taylor J, Nowozin S (2016) Learning step size controllers for robust neural network training. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 30","DOI":"10.1609\/aaai.v30i1.10187"},{"key":"2226_CR24","unstructured":"Xu C, Qin T, Wang G, Liu T-Y (2017) Reinforcement learning for learning rate control. arXiv preprint arXiv:1705.11159"},{"key":"2226_CR25","unstructured":"Xu Z, Dai AM, Kemp J, Metz L (2019) Learning an adaptive learning rate schedule. arXiv preprint arXiv:1909.09712"},{"key":"2226_CR26","unstructured":"Andrychowicz M, Denil M, Gomez S, Hoffman MW, Pfau D, Schaul T, Shillingford B, De\u00a0Freitas N (2016) Learning to learn by gradient descent by gradient descent. Advances in neural information processing systems 29"},{"key":"2226_CR27","doi-asserted-by":"crossref","unstructured":"Egidio LN, Hansson A, Wahlberg B (2021) Learning the step-size policy for the limited-memory broyden-fletcher-goldfarb-shanno algorithm. In: 2021 International Joint Conference on Neural Networks (IJCNN), pp. 1\u20138. IEEE","DOI":"10.1109\/IJCNN52387.2021.9534194"},{"issue":"3","key":"2226_CR28","doi-asserted-by":"publisher","first-page":"3505","DOI":"10.1109\/TPAMI.2022.3184315","volume":"45","author":"J Shu","year":"2022","unstructured":"Shu J, Zhu Y, Zhao Q, Meng D, Xu Z (2022) Mlr-snet: Transferable lr schedules for heterogeneous tasks. IEEE Trans Pattern Anal Mach Intell 45(3):3505\u20133521","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"2226_CR29","unstructured":"Franceschi L, Donini M, Frasconi P, Pontil M (2017) Forward and reverse gradient-based hyperparameter optimization. In: International Conference on Machine Learning, pp. 1165\u20131173. PMLR"},{"key":"2226_CR30","unstructured":"Baydin AG, Cornish R, Rubio DM, Schmidt M, Wood F (2018) Online learning rate adaptation with hypergradient descent. In: International Conference on Learning Representations"},{"key":"2226_CR31","doi-asserted-by":"crossref","unstructured":"Donini M, Franceschi L, Majumder O, Pontil M, Frasconi P (2021) Marthe: scheduling the learning rate via online hypergradients. In: Proceedings of the Twenty-Ninth International Conference on International Joint Conferences on Artificial Intelligence, pp. 2119\u20132125","DOI":"10.24963\/ijcai.2020\/293"},{"key":"2226_CR32","doi-asserted-by":"publisher","first-page":"11","DOI":"10.1016\/j.neunet.2021.03.025","volume":"141","author":"Y Zhu","year":"2021","unstructured":"Zhu Y, Huang D, Gao Y, Wu R, Chen Y, Zhang B, Wang H (2021) Automatic, dynamic, and nearly optimal learning rate specification via local quadratic approximation. Neural Netw 141:11\u201329","journal-title":"Neural Netw"},{"key":"2226_CR33","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2021.115357","volume":"184","author":"Y Li","year":"2021","unstructured":"Li Y, Zhang Q, Yoon SW (2021) Gaussian process regression-based learning rate optimization in convolutional neural networks for medical images classification. Expert Syst Appl 184:115357","journal-title":"Expert Syst Appl"},{"issue":"7","key":"2226_CR34","doi-asserted-by":"publisher","first-page":"6739","DOI":"10.1007\/s11071-024-10449-6","volume":"113","author":"G Yang","year":"2025","unstructured":"Yang G (2025) State filtered disturbance rejection control. Nonlinear Dyn 113(7):6739\u20136755","journal-title":"Nonlinear Dyn"},{"issue":"4","key":"2226_CR35","doi-asserted-by":"publisher","first-page":"2972","DOI":"10.1002\/rnc.7118","volume":"34","author":"G Yang","year":"2024","unstructured":"Yang G, Yao J (2024) Multilayer neurocontrol of high-order uncertain nonlinear systems with active disturbance rejection. Int J Robust Nonlinear Control 34(4):2972\u20132987","journal-title":"Int J Robust Nonlinear Control"},{"key":"2226_CR36","unstructured":"Krizhevsky A, et al (2009) Learning multiple layers of features from tiny images"},{"key":"2226_CR37","unstructured":"Le Y, Yang X (2015) Tiny imagenet visual recognition challenge. CS 231N 7(7), 3"},{"issue":"3","key":"2226_CR38","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky O, Deng J, Su H, Krause J, Satheesh S, Ma S, Huang Z, Karpathy A, Khosla A, Bernstein M et al (2015) Imagenet large scale visual recognition challenge. Int J Comput Vision 115(3):211\u2013252","journal-title":"Int J Comput Vision"}],"container-title":["Complex &amp; Intelligent Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s40747-026-02226-2","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s40747-026-02226-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s40747-026-02226-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T07:55:10Z","timestamp":1774425310000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s40747-026-02226-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,31]]},"references-count":38,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,3]]}},"alternative-id":["2226"],"URL":"https:\/\/doi.org\/10.1007\/s40747-026-02226-2","relation":{},"ISSN":["2199-4536","2198-6053"],"issn-type":[{"value":"2199-4536","type":"print"},{"value":"2198-6053","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1,31]]},"assertion":[{"value":"17 September 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 December 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 January 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no conflict of interest.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"106"}}