{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,25]],"date-time":"2026-05-25T07:05:32Z","timestamp":1779692732147,"version":"3.53.1"},"reference-count":46,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neural Networks"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.neunet.2026.108799","type":"journal-article","created":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T15:00:32Z","timestamp":1772377232000},"page":"108799","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Stabilizing iterative pruning with local adjustments and global scaling learning rate"],"prefix":"10.1016","volume":"200","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-2446-2089","authenticated-orcid":false,"given":"Lian","family":"Duan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4549-814X","authenticated-orcid":false,"given":"Jiawen","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-0597-9026","authenticated-orcid":false,"given":"Chongxin","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4649-7361","authenticated-orcid":false,"given":"Hanzhang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neunet.2026.108799_bib0001","series-title":"2017 innovations in power and advanced computing technologies (i-PACT)","first-page":"1","article-title":"Improvement in the performance of deep neural network model using learning rate","author":"Arora","year":"2017"},{"key":"10.1016\/j.neunet.2026.108799_bib0002","unstructured":"Bohnstingl, T., Garg, A., Wo\u017aniak, S., Saon, G., Eleftheriou, E., & Pantazi, A. (2021). Towards efficient end-to-end speech recognition with biologically-inspired neural networks. arXiv preprint arXiv: 2110.02743."},{"key":"10.1016\/j.neunet.2026.108799_bib0003","first-page":"15834","article-title":"The lottery ticket hypothesis for pre-trained bert networks","volume":"33","author":"Chen","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"12","key":"10.1016\/j.neunet.2026.108799_bib0004","doi-asserted-by":"crossref","first-page":"10558","DOI":"10.1109\/TPAMI.2024.3447085","article-title":"A survey on deep neural network pruning: Taxonomy, comparison, analysis, and recommendations","volume":"46","author":"Cheng","year":"2024","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.neunet.2026.108799_bib0005","series-title":"2009\u202fIEEE conference on computer vision and pattern recognition","first-page":"248","article-title":"Imagenet: A large-scale hierarchical image database","author":"Deng","year":"2009"},{"key":"10.1016\/j.neunet.2026.108799_bib0006","unstructured":"Devlin, J., Chang, M.-W., Lee, K., & Bert, K. T. (2018). Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv: 1810.04805."},{"key":"10.1016\/j.neunet.2026.108799_bib0007","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S. et al. (2020). An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv: 2010.11929."},{"key":"10.1016\/j.neunet.2026.108799_bib0008","first-page":"2121","article-title":"Adaptive subgradient methods for online learning and stochastic optimization","volume":"12","author":"Duchi","year":"2011","journal-title":"Journal of Machine Learning Research"},{"key":"10.1016\/j.neunet.2026.108799_bib0009","series-title":"International conference on machine learning","first-page":"10323","article-title":"Sparsegpt: Massive language models can be accurately pruned in one-shot","author":"Frantar","year":"2023"},{"key":"10.1016\/j.neunet.2026.108799_bib0010","unstructured":"Han, S., Mao, H., & Dally, W. J. (2015a). Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding. arXiv preprint arXiv: 1510.00149."},{"key":"10.1016\/j.neunet.2026.108799_bib0011","article-title":"Learning both weights and connections for efficient neural network","volume":"28","author":"Han","year":"2015","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.108799_bib0012","article-title":"Control batch size and learning rate to generalize well: Theoretical and empirical evidence","volume":"32","author":"He","year":"2019","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.108799_bib0013","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2015). Deep residual learning for image recognition. https:\/\/arxiv.org\/abs\/1512.03385."},{"key":"10.1016\/j.neunet.2026.108799_bib0014","unstructured":"He, P., Liu, X., Gao, J., & Chen, W. (2020). Deberta: Decoding-enhanced bert with disentangled attention. arXiv preprint arXiv: 2006.03654."},{"key":"10.1016\/j.neunet.2026.108799_bib0015","unstructured":"Hinton, G., Srivastava, N., & Swersky, K. (2010). Neural networks for machine learning, lecture 6a: Overview of mini-batch gradient descent. Lecture notes distributed in CSC321, University of Toronto."},{"issue":"4","key":"10.1016\/j.neunet.2026.108799_bib0016","doi-asserted-by":"crossref","first-page":"295","DOI":"10.1016\/0893-6080(88)90003-2","article-title":"Increased rates of convergence through learning rate adaptation","volume":"1","author":"Jacobs","year":"1988","journal-title":"Neural Networks"},{"key":"10.1016\/j.neunet.2026.108799_bib0017","series-title":"Technical Report","article-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky","year":"2009"},{"key":"10.1016\/j.neunet.2026.108799_bib0018","unstructured":"Le, D. H., & Hua, B.-S. (2021). Network pruning that matters: A case study on retraining variants. arXiv preprint arXiv: 2105.03193."},{"key":"10.1016\/j.neunet.2026.108799_bib0019","unstructured":"Lee, N., Ajanthan, T., & Torr, P. H. S. (2018). Snip: Single-shot network pruning based on connection sensitivity. arXiv preprint arXiv: 1810.02340."},{"key":"10.1016\/j.neunet.2026.108799_bib0020","unstructured":"Liang, C., Jiang, H., Zuo, S., He, P., Liu, X., Gao, J., Chen, W., & Zhao, T. (2022). No parameters left behind: Sensitivity guided adaptive learning rate for training large transformer models. arXiv preprint arXiv: 2202.02664."},{"key":"10.1016\/j.neunet.2026.108799_bib0021","unstructured":"Liang, C., Zuo, S., Chen, M., Jiang, H., Liu, X., He, P., Zhao, T., & Chen, W. (2021). Super tickets in pre-trained language models: From model compression to improving generalization. arXiv preprint arXiv: 2105.12002."},{"key":"10.1016\/j.neunet.2026.108799_bib0022","unstructured":"Liu, Y., Ott, M., Goyal, N., Du, J., Joshi, M., Chen, D., Levy, O., Lewis, M., Zettlemoyer, L., & Stoyanov, V. (2019). Roberta: A robustly optimized BERT pretraining approach. arXiv preprint arXiv: 1907.11692."},{"key":"10.1016\/j.neunet.2026.108799_bib0023","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"10012","article-title":"Swin transformer: Hierarchical vision transformer using shifted windows","author":"Liu","year":"2021"},{"key":"10.1016\/j.neunet.2026.108799_bib0024","unstructured":"Loshchilov, I., & Hutter, F. (2017). Decoupled weight decay regularization. arXiv preprint arXiv: 1711.05101."},{"key":"10.1016\/j.neunet.2026.108799_bib0025","unstructured":"Luo, L., Xiong, Y., Liu, Y., & Sun, X. (2019). Adaptive gradient methods with dynamic bound of learning rate. arXiv preprint arXiv: 1902.09843."},{"key":"10.1016\/j.neunet.2026.108799_bib0026","first-page":"21702","article-title":"Llm-pruner: On the structural pruning of large language models","volume":"36","author":"Ma","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.108799_bib0027","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"11264","article-title":"Importance estimation for neural network pruning","author":"Molchanov","year":"2019"},{"key":"10.1016\/j.neunet.2026.108799_bib0028","doi-asserted-by":"crossref","first-page":"11","DOI":"10.1016\/j.neucom.2018.01.046","article-title":"A modified elman neural network with a new learning rate scheme","volume":"286","author":"Ren","year":"2018","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neunet.2026.108799_bib0029","unstructured":"Renda, A., Frankle, J., & Carbin, M. (2020). Comparing rewinding and fine-tuning in neural network pruning. arXiv preprint arXiv: 2003.02389."},{"key":"10.1016\/j.neunet.2026.108799_bib0030","series-title":"35th AAAI conference on artificial intelligence, AAAI 2021","first-page":"2486","article-title":"AutoLR: Layer-wise pruning and auto-tuning of learning rates in fine-tuning of deep networks","author":"Ro","year":"2021"},{"key":"10.1016\/j.neunet.2026.108799_bib0031","series-title":"Advances in neural information processing systems","first-page":"20378","article-title":"Movement pruning: Adaptive sparsity by fine-tuning","volume":"vol. 33","author":"Sanh","year":"2020"},{"key":"10.1016\/j.neunet.2026.108799_bib0032","series-title":"Artificial intelligence and machine learning for multi-domain operations applications","first-page":"369","article-title":"Super-convergence: Very fast training of neural networks using large learning rates","volume":"vol. 11006","author":"Smith","year":"2019"},{"key":"10.1016\/j.neunet.2026.108799_bib0033","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"2762","article-title":"Cyclical pruning for sparse neural networks","author":"Srinivas","year":"2022"},{"key":"10.1016\/j.neunet.2026.108799_bib0034","series-title":"International conference on machine learning","first-page":"10347","article-title":"Training data-efficient image transformers & distillation through attention","author":"Touvron","year":"2021"},{"key":"10.1016\/j.neunet.2026.108799_bib0035","doi-asserted-by":"crossref","first-page":"63280","DOI":"10.1109\/ACCESS.2022.3182659","article-title":"Methods for pruning deep neural networks","volume":"10","author":"Vadera","year":"2022","journal-title":"IEEE Access"},{"key":"10.1016\/j.neunet.2026.108799_bib0036","unstructured":"Wang, H., Qin, C., Bai, Y., & Fu, Y. (2023). Why is the state of neural network pruning so confusing? On the fairness, comparison setup, and trainability in network pruning. arXiv preprint arXiv: 2301.05219."},{"key":"10.1016\/j.neunet.2026.108799_bib0037","series-title":"Proceedings of the 41st international conference on machine learning","first-page":"52247","article-title":"Exploring intrinsic dimension for vision-language model pruning","volume":"vol. 235","author":"Wang","year":"2024"},{"issue":"5","key":"10.1016\/j.neunet.2026.108799_bib0038","doi-asserted-by":"crossref","first-page":"7060","DOI":"10.1109\/TNNLS.2022.3213677","article-title":"Incremental PID controller-based learning rate scheduler for stochastic gradient descent","volume":"35","author":"Wang","year":"2024","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.neunet.2026.108799_bib0039","unstructured":"Wu, J., Gan, W., Chen, Z., Wan, S., & Lin, H. (2023). Ai-generated content (aigc): A survey. arXiv preprint arXiv: 2304.06632."},{"issue":"12","key":"10.1016\/j.neunet.2026.108799_bib0040","doi-asserted-by":"crossref","first-page":"7197","DOI":"10.1109\/TCSVT.2023.3277689","article-title":"Skeleton neural networks via low-rank guided filter pruning","volume":"33","author":"Yang","year":"2023","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.neunet.2026.108799_bib0041","article-title":"How transferable are features in deep neural networks?","volume":"27","author":"Yosinski","year":"2014","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"9","key":"10.1016\/j.neunet.2026.108799_bib0042","doi-asserted-by":"crossref","first-page":"12518","DOI":"10.1109\/TNNLS.2023.3263393","article-title":"Salr: Sharpness-aware learning rate scheduler for improved generalization","volume":"35","author":"Yue","year":"2024","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.neunet.2026.108799_bib0043","series-title":"Computer vision\u2013ECCV 2014: 13th European conference, Zurich, Switzerland, September 6\u201312, 2014, proceedings, part I 13","first-page":"818","article-title":"Visualizing and understanding convolutional networks","author":"Zeiler","year":"2014"},{"key":"10.1016\/j.neunet.2026.108799_bib0044","series-title":"International conference on machine learning","first-page":"26809","article-title":"Platon: Pruning large transformer models with upper confidence bound of weight importance","author":"Zhang","year":"2022"},{"issue":"8","key":"10.1016\/j.neunet.2026.108799_bib0045","doi-asserted-by":"crossref","first-page":"10996","DOI":"10.1109\/TNNLS.2023.3246263","article-title":"Adaptive filter pruning via sensitivity feedback","volume":"35","author":"Zhang","year":"2024","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.neunet.2026.108799_bib0046","unstructured":"Zhu, M., & Gupta, S. (2017). To prune, or not to prune: Exploring the efficacy of pruning for model compression. arXiv preprint arXiv: 1710.01878."}],"container-title":["Neural Networks"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026002613?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026002613?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,25]],"date-time":"2026-05-25T06:37:57Z","timestamp":1779691077000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0893608026002613"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":46,"alternative-id":["S0893608026002613"],"URL":"https:\/\/doi.org\/10.1016\/j.neunet.2026.108799","relation":{},"ISSN":["0893-6080"],"issn-type":[{"value":"0893-6080","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Stabilizing iterative pruning with local adjustments and global scaling learning rate","name":"articletitle","label":"Article Title"},{"value":"Neural Networks","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neunet.2026.108799","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"108799"}}