{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T16:41:56Z","timestamp":1783615316188,"version":"3.55.0"},"reference-count":77,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"8","license":[{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Italian Ministry of Defence","award":["CIG: Z84333EA0D"],"award-info":[{"award-number":["CIG: Z84333EA0D"]}]},{"DOI":"10.13039\/100031478","name":"NextGenerationEU","doi-asserted-by":"publisher","award":["PE00000013"],"award-info":[{"award-number":["PE00000013"]}],"id":[{"id":"10.13039\/100031478","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Artif. Intell."],"published-print":{"date-parts":[[2024,8]]},"DOI":"10.1109\/tai.2024.3366497","type":"journal-article","created":{"date-parts":[[2024,2,15]],"date-time":"2024-02-15T14:16:03Z","timestamp":1708006563000},"page":"4269-4279","source":"Crossref","is-referenced-by-count":18,"title":["Distilled Gradual Pruning With Pruned Fine-Tuning"],"prefix":"10.1109","volume":"5","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-0437-7832","authenticated-orcid":false,"given":"Federico","family":"Fontana","sequence":"first","affiliation":[{"name":"Department of Computer Science, Sapienza University of Rome, Rome, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2939-3007","authenticated-orcid":false,"given":"Romeo","family":"Lanzino","sequence":"additional","affiliation":[{"name":"Department of Computer Science, Sapienza University of Rome, Rome, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2540-2570","authenticated-orcid":false,"given":"Marco Raoul","family":"Marini","sequence":"additional","affiliation":[{"name":"Department of Computer Science, Sapienza University of Rome, Rome, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9437-6217","authenticated-orcid":false,"given":"Danilo","family":"Avola","sequence":"additional","affiliation":[{"name":"Department of Computer Science, Sapienza University of Rome, Rome, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9149-2175","authenticated-orcid":false,"given":"Luigi","family":"Cinque","sequence":"additional","affiliation":[{"name":"Department of Computer Science, Sapienza University of Rome, Rome, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7765-1563","authenticated-orcid":false,"given":"Francesco","family":"Scarcello","sequence":"additional","affiliation":[{"name":"Department of Computer Engineering, Modeling, Electronics and Systems, University of Calabria, Rende, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8425-6892","authenticated-orcid":false,"given":"Gian Luca","family":"Foresti","sequence":"additional","affiliation":[{"name":"Department of Mathematics, Computer Science and Physics, University of Udine, Udine, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"1","article-title":"ImageNet classification with deep convolutional neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"25","author":"Krizhevsky","year":"2012"},{"key":"ref2","first-page":"1","article-title":"Very deep convolutional networks for large-scale image recognition","volume-title":"Proc. Int. Conf. Learn. Representations (ICLR)","author":"Simonyan","year":"2015"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2016.90"},{"key":"ref4","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2016.91"},{"key":"ref6","first-page":"1","article-title":"Faster R-CNN: Towards real-time object detection with region proposal networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"28","author":"Ren","year":"2015"},{"key":"ref7","article-title":"Insta-yolo: Real-time instance segmentation","author":"Mohamed","year":"2021"},{"key":"ref8","first-page":"2980","article-title":"Mask R-CNN","volume-title":"Proc. IEEE Int. Conf. Comput. Vis. (ICCV)","author":"He","year":"2017"},{"key":"ref9","first-page":"10734","article-title":"FBNet: Hardware-aware efficient ConvNet design via differentiable neural architecture search","volume-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","author":"Wu","year":"2019"},{"key":"ref10","first-page":"3822","article-title":"Breaking the curse of space explosion: Towards efficient NAS with curriculum search","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Guo","year":"2020"},{"key":"ref11","first-page":"9502","article-title":"Contrastive neural architecture search with neural architecture comparators","volume-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit.","author":"Chen","year":"2021"},{"key":"ref12","article-title":"Once-for-all: Train one network and specialize it for efficient deployment","author":"Cai","year":"2019"},{"issue":"10","key":"ref13","doi-asserted-by":"crossref","first-page":"6501","DOI":"10.1109\/TPAMI.2021.3086914","article-title":"Towards accurate and compact architectures via neural architecture transformer","volume":"44","author":"Guo","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"1","key":"ref14","doi-asserted-by":"crossref","first-page":"553","DOI":"10.1016\/j.neunet.2021.09.002","article-title":"Disturbance-immune weight sharing for neural architecture search","volume":"144","author":"Niu","year":"2021","journal-title":"Neural Netw."},{"key":"ref15","first-page":"770","article-title":"Deep residual learning for image recognition","volume-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR)","author":"He","year":"2016"},{"key":"ref16","first-page":"1","article-title":"Pruning filters for efficient ConvNets","volume-title":"Int. Conf. Learn. Representations (ICLR)","author":"Li","year":"2017"},{"key":"ref17","first-page":"9270","article-title":"Network pruning via performance maximization","volume-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","author":"Gao","year":"2021"},{"key":"ref18","first-page":"4510","article-title":"ResRep: Lossless CNN pruning via decoupling remembering and forgetting","volume-title":"Proc. IEEE\/CVF Int. Conf. Comput. Vis.","author":"Ding","year":"2021"},{"issue":"8","key":"ref19","first-page":"4035","article-title":"Discrimination-aware network pruning for deep model compression","volume":"44","author":"Liu","year":"2021","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"1","key":"ref20","doi-asserted-by":"crossref","first-page":"370","DOI":"10.1016\/j.neucom.2021.07.045","article-title":"Pruning and quantization for deep neural network acceleration: A survey","volume":"461","author":"Liang","year":"2021","journal-title":"Neurocomputing"},{"issue":"9","key":"ref21","doi-asserted-by":"crossref","first-page":"4930","DOI":"10.1109\/TNNLS.2021.3063265","article-title":"Non-structured DNN weight pruning\u2014Is it beneficial in any platform?","volume":"33","author":"Ma","year":"2021","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"ref22","first-page":"1","article-title":"Accelerating sparse deep neural networks on FPGAS","volume-title":"Proc. IEEE High Perform. Extreme Comput. Conf. (HPEC)","author":"Huang","year":"2019"},{"issue":"1","key":"ref23","doi-asserted-by":"crossref","first-page":"47","DOI":"10.1109\/TSUSC.2021.3060690","article-title":"AdaPrune: An accelerator-aware pruning technique for sustainable CNN accelerators","volume":"7","author":"Li","year":"2022","journal-title":"IEEE Trans. Sustain. Comput."},{"key":"ref24","first-page":"598","article-title":"Optimal brain damage","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"LeCun","year":"1990"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/3144789.3144803"},{"key":"ref26","article-title":"Pruning convolutional neural networks for resource efficient inference","author":"Molchanov","year":"2016"},{"key":"ref27","article-title":"To prune, or not to prune: Exploring the efficacy of pruning for model compression","author":"Zhu","year":"2017"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/cvprw56347.2022.00312"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.00029"},{"key":"ref30","article-title":"FitNets: Hints for thin deep nets","author":"Romero","year":"2014"},{"issue":"1","key":"ref31","article-title":"Data-free knowledge distillation in neural networks for regression","volume":"175","author":"Kang","year":"2021","journal-title":"Expert Syst. Appl."},{"issue":"8","key":"ref32","first-page":"4388","article-title":"Self-distillation: Towards efficient and compact neural networks","volume":"44","author":"Zhang","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"ref33","first-page":"1398","article-title":"Channel pruning for accelerating very deep neural networks","volume-title":"Int. Conf. Comput. Vis. (ICCV)","author":"He","year":"2017"},{"key":"ref34","first-page":"1","article-title":"NVIDIA A100 GPU: Performance & innovation for GPU computing","volume-title":"Proc. IEEE Hot Chips 32 Symp. (HCS)","author":"Choquette","year":"2020"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-11021-5_19"},{"issue":"8","key":"ref36","first-page":"4035","article-title":"Discrimination-aware network pruning for deep model compression","volume":"44","author":"Liu","year":"2021","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"ref37","first-page":"1135","article-title":"Learning both weights and connections for efficient neural networks","volume-title":"Proc. 28th Int. Conf. Neural Inf. Process. Syst.","volume":"1","author":"Han","year":"2015"},{"key":"ref38","first-page":"10,217","article-title":"Parameter-efficient masking networks","volume":"35","author":"Bai","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref39","article-title":"A unified framework for soft threshold pruning","author":"Chen","year":"2023"},{"key":"ref40","article-title":"Soft threshold weight reparameterization for learnable sparsity","volume-title":"Proc. Int. Conf. Mach. Learn. (UCML)","author":"Kusupati","year":"2020"},{"key":"ref41","article-title":"Learned threshold pruning","author":"Azarian","year":"2020"},{"key":"ref42","first-page":"164","article-title":"Second order derivatives for network pruning: Optimal brain surgeon","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Hassibi","year":"1993"},{"key":"ref43","article-title":"Faster gaze prediction with dense networks and fisher pruning","author":"Theis","year":"2018"},{"key":"ref44","first-page":"1","article-title":"WoodFisher: Efficient second-order approximations for model compression","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Singh","year":"2020"},{"key":"ref45","article-title":"Revisiting loss modelling for unstructured pruning","author":"Laurent","year":"2021"},{"key":"ref46","first-page":"3288","article-title":"Bayesian compression for deep learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Louizos","year":"2017"},{"key":"ref47","first-page":"1135","article-title":"Compressing neural networks using the variational information bottleneck","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Dai","year":"2018"},{"key":"ref48","article-title":"The state of sparsity in deep neural networks","author":"Gale","year":"2019"},{"key":"ref49","first-page":"2943","article-title":"Rigging the lottery: Making all tickets winners","volume-title":"Proc. 37th Int. Conf. Mach. Learn. (ICML)","author":"Evci","year":"2020"},{"key":"ref50","first-page":"20, 744","article-title":"Top-KAST: Top-K always sparse training","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Jayakumar","year":"2020"},{"key":"ref51","first-page":"1387","article-title":"Dynamic network surgery for efficient DNNs","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Guo","year":"2016"},{"key":"ref52","first-page":"1","article-title":"Dynamic model pruning with feedback","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Lin","year":"2020"},{"key":"ref53","article-title":"Distilling the knowledge in a neural network","author":"Hinton","year":"2015"},{"key":"ref54","first-page":"7130","article-title":"A gift from knowledge distillation: Fast optimization, network minimization and transfer learning","volume-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR)","author":"Yim","year":"2017"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20083-0_8"},{"key":"ref56","first-page":"3191","article-title":"Combining weight pruning and knowledge distillation for CNN compression","volume-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR) Workshops","author":"Aghli","year":"2021"},{"issue":"10","key":"ref57","doi-asserted-by":"crossref","first-page":"7350","DOI":"10.1109\/TNNLS.2022.3141665","article-title":"Automatic sparse connectivity learning for neural networks","volume":"34","author":"Tang","year":"2023","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"ref58","first-page":"1","article-title":"A fair loss function for network pruning","volume-title":"Proc. Workshop Trustworthy Socially Responsible Mach. Learn. (TSRML)","author":"Meyer","year":"2022"},{"key":"ref59","doi-asserted-by":"crossref","DOI":"10.24963\/ijcai.2021\/362","article-title":"Comparing Kullback-Leibler divergence and mean squared error loss in knowledge distillation","author":"Kim","year":"2021"},{"key":"ref60","article-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2017"},{"key":"ref61","first-page":"32","article-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky","year":"2009"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"ref63","first-page":"4128","article-title":"One-cycle pruning: Pruning ConvNets with tight training budget","volume-title":"Proc. IEEE Int. Conf. Image Process. (ICIP)","author":"Hubens","year":"2022"},{"key":"ref64","article-title":"Snip: Single-shot network pruning based on connection sensitivity","author":"Lee","year":"2018"},{"key":"ref65","first-page":"129","article-title":"What is the state of neural network pruning?","volume-title":"Proc. Mach. Learn. Syst.","volume":"2","author":"Blalock","year":"2020"},{"key":"ref66","first-page":"9908","article-title":"Sparse training via boosting pruning plasticity with neuroregeneration","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Liu","year":"2021"},{"key":"ref67","article-title":"OptG: Optimizing gradient-driven criteria in network sparsity","author":"Zhang","year":"2022"},{"key":"ref68","first-page":"4510","article-title":"MobileNetV2: Inverted residuals and linear bottlenecks","volume-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR)","author":"Sandler","year":"2018"},{"issue":"3","key":"ref69","first-page":"1","article-title":"Rethinking weight decay for efficient neural network pruning","volume":"8","author":"Tessier","year":"2022","journal-title":"J. Imag."},{"key":"ref70","first-page":"5544","article-title":"Soft threshold weight reparameterization for learnable sparsity","volume-title":"Proc. 37th Int. Conf. Mach. Learn. (ICML)","author":"Kusupati","year":"2020"},{"key":"ref71","article-title":"MLPrune: Multi-layer pruning for automated neural network compression","author":"Zeng","year":"2019"},{"key":"ref72","article-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and Huffman coding","author":"Han","year":"2015"},{"key":"ref73","article-title":"To prune, or not to prune: Exploring the efficacy of pruning for model compression","author":"Zhu","year":"2017"},{"key":"ref74","first-page":"1","article-title":"Discovering neural wirings","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Wortsman","year":"2019"},{"key":"ref75","first-page":"19, 974","article-title":"Chasing sparsity in vision transformers: An end-to-end exploration","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Chen","year":"2021"},{"issue":"3","key":"ref76","first-page":"3143","article-title":"Width & depth pruning for vision transformers","volume-title":"Proc. AAAI Conf. Artif. Intell.","volume":"36","author":"Yu","year":"2022"},{"key":"ref77","first-page":"24355","article-title":"X-Pruner: Explainable pruning for vision transformers","volume-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","author":"Yu","year":"2023"}],"container-title":["IEEE Transactions on Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9078688\/10635096\/10438214.pdf?arnumber=10438214","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T01:09:03Z","timestamp":1755911343000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10438214\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8]]},"references-count":77,"journal-issue":{"issue":"8"},"URL":"https:\/\/doi.org\/10.1109\/tai.2024.3366497","relation":{},"ISSN":["2691-4581"],"issn-type":[{"value":"2691-4581","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,8]]}}}