{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,4]],"date-time":"2026-06-04T16:04:53Z","timestamp":1780589093647,"version":"3.54.1"},"reference-count":59,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2023,3,14]],"date-time":"2023-03-14T00:00:00Z","timestamp":1678752000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,3,14]],"date-time":"2023-03-14T00:00:00Z","timestamp":1678752000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2023,7]]},"DOI":"10.1007\/s10994-023-06304-1","type":"journal-article","created":{"date-parts":[[2023,3,14]],"date-time":"2023-03-14T21:28:56Z","timestamp":1678829336000},"page":"2653-2684","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Pruning during training by network efficacy modeling"],"prefix":"10.1007","volume":"112","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8928-6302","authenticated-orcid":false,"given":"Mohit","family":"Rajpal","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yehong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bryan Kian Hsiang","family":"Low","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,3,14]]},"reference":[{"key":"6304_CR1","unstructured":"Allen-Zhu, Z., Li, Y., & Liang, Y. (2019). Learning and generalization in overparameterized neural networks, going beyond two layers. In Proceedings of the NeurIPS (pp. 6155\u20136166)."},{"issue":"1","key":"6304_CR2","first-page":"1459","volume":"12","author":"MA \u00c1lvarez","year":"2011","unstructured":"\u00c1lvarez, M. A., & Lawrence, N. D. (2011). Computationally efficient convolved multiple output Gaussian processes. JMLR, 12(1), 1459\u20131500.","journal-title":"JMLR"},{"issue":"4","key":"6304_CR3","doi-asserted-by":"publisher","first-page":"699","DOI":"10.1016\/0967-0661(93)91394-C","volume":"1","author":"KJ \u00c5str\u00f6m","year":"1993","unstructured":"\u00c5str\u00f6m, K. J., H\u00e4gglund, T., Hang, C. C., & Ho, W. K. (1993). Automatic tuning and adaptation for PID controllers: A survey. Control Engineering Practice, 1(4), 699\u2013714.","journal-title":"Control Engineering Practice"},{"key":"6304_CR4","unstructured":"Bellec, G., Kappel, D., Maass, W., & Legenstein, R. A. (2018). Deep rewiring: Training very sparse deep networks. In Proceedings of the ICLR."},{"key":"6304_CR5","volume-title":"Adaptive control processes: A guided tour","author":"RE Bellman","year":"2015","unstructured":"Bellman, R. E. (2015). Adaptive control processes: A guided tour. Princeton University Press."},{"key":"6304_CR6","doi-asserted-by":"crossref","unstructured":"Bulu\u00e7, A. , & Gilbert, J. R. (2008). Challenges and advances in parallel sparse matrix-matrix multiplication. In Proceedings of the ICCP (pp. 503\u2013510).","DOI":"10.1109\/ICPP.2008.45"},{"key":"6304_CR7","unstructured":"Courbariaux, M., Bengio, Y., & David, J. (2015). BinaryConnect: Training deep neural networks with binary weights during propagations. arXiv:1511.00363."},{"issue":"10","key":"6304_CR8","doi-asserted-by":"publisher","first-page":"1487","DOI":"10.1109\/TC.2019.2914438","volume":"68","author":"X Dai","year":"2019","unstructured":"Dai, X., Yin, H., & Jha, N. K. (2019). Nest: A neural network synthesis tool based on a grow-and-prune paradigm. IEEE Transactions on Computers, 68(10), 1487\u20131497.","journal-title":"IEEE Transactions on Computers"},{"key":"6304_CR9","unstructured":"de Jorge, P., Sanyal, A., Behl, H. S., Torr, P. H. S., Rogez, G., & Dokania, P. K. (2021). Progressive skeletonization: Trimming more fat from a network at initialization. In Proceedings of the ICLR."},{"key":"6304_CR10","unstructured":"Denton, E. L., Zaremba, W., Bruna, J., LeCun, Y., Fergus, R. (2014). Exploiting linear structure within convolutional networks for efficient evaluation. In Proceedings of the NeurIPS (pp. 1269\u20131277)."},{"key":"6304_CR11","unstructured":"Dettmers, T., & Zettlemoyer, L. (2019). Sparse networks from scratch: Faster training without losing performance. arXiv:1907.04840."},{"key":"6304_CR12","unstructured":"Dong, X., Chen, S., & Pan, S. J. (2017). Learning to prune deep neural networks via layer-wise optimal brain surgeon. In Proceedings of the NeurIPS (pp. 4857\u20134867)."},{"key":"6304_CR13","unstructured":"Frankle, J., & Carbin, M. (2019). The lottery ticket hypothesis: Finding sparse, trainable neural networks. In Proceedings of the ICLR."},{"key":"6304_CR14","unstructured":"Gale, T., Elsen, E., & Hooker, S. (2019). The state of sparsity in deep neural networks. arXiv:1902.09574."},{"key":"6304_CR15","unstructured":"Guo, Y., Yao, A., & Chen, Y. (2016). Dynamic network surgery for efficient DNNs. In Proceedings of the NeurIPS (pp. 1379\u20131387)."},{"key":"6304_CR16","unstructured":"Han, S., Pool, J., Tran, J., & Dally, W. (2015). Learning both weights and connections for efficient neural networks. In Proceedings of the NeurIPS (pp. 1135\u20131143)."},{"key":"6304_CR17","unstructured":"Hassibi, B., & Stork, D. G. (1992). Second order derivatives for network pruning: Optimal brain surgeon. In Proceedings of the NeurIPS (pp. 164\u2013171)."},{"key":"6304_CR18","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016a). Deep residual learning for image recognition. In Proceedings of the CVPR (pp. 770\u2013778).","DOI":"10.1109\/CVPR.2016.90"},{"key":"6304_CR19","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016b). Identity mappings in deep residual networks. In Proceedings of the ECCV (pp. 4432\u20134440).","DOI":"10.1007\/978-3-319-46493-0_38"},{"key":"6304_CR20","doi-asserted-by":"crossref","unstructured":"He, Y., Lin, J., Liu, Z., Wang, H., Li, L., & Han, S. (2018). AMC: AutoML for model compression and acceleration on mobile devices. In Proceedings of the ECCV (pp. 784\u2013800).","DOI":"10.1007\/978-3-030-01234-2_48"},{"key":"6304_CR21","unstructured":"Hensman, J., Matthews, A., & Ghahramani, Z. (2015). Scalable variational Gaussian process classification. In Proceedings of the AISTATS (pp. 351\u2013360)."},{"key":"6304_CR22","unstructured":"Hinton, G. E. , Srivastava, N., Krizhevsky, A., Sutskever, I., & Salakhutdinov, R. R. (2012). Improving neural networks by preventing co-adaptation of feature detectors. arXiv:1207.0580."},{"key":"6304_CR23","unstructured":"Hinton, G. E., Vinyals, O., & Dean, J. (2015). Distilling the knowledge in a neural network. arXiv:1503.02531."},{"issue":"1","key":"6304_CR24","first-page":"6869","volume":"18","author":"I Hubara","year":"2017","unstructured":"Hubara, I., Courbariaux, M., Soudry, D., El-Yaniv, R., & Bengio, Y. (2017). Quantized neural networks: Training neural networks with low precision weights and activations. JMLR, 18(1), 6869\u20136898.","journal-title":"JMLR"},{"key":"6304_CR25","doi-asserted-by":"crossref","unstructured":"Idelbayev, Y., & Carreira-Perpi\u00f1\u00e1n, M. \u00c1. (2021a). LC: A flexible, extensible open-source toolkit for model compression. In Proceedings of the CIKM (pp. 4504\u20134514).","DOI":"10.1145\/3459637.3482005"},{"key":"6304_CR26","doi-asserted-by":"crossref","unstructured":"Idelbayev, Y., & Carreira-Perpi\u00f1\u00e1n, M. \u00c1. (2021b). More general and effective model compression via an additive combination of compressions. In Proceedings of the ECML PKDD research track (Vol. 12977, pp. 233\u2013248). Springer.","DOI":"10.1007\/978-3-030-86523-8_15"},{"key":"6304_CR27","doi-asserted-by":"crossref","unstructured":"Jaderberg, M., Vedaldi, A., & Zisserman, A. (2014). Speeding up convolutional neural networks with low rank expansions. Proceedings of the BMVC.","DOI":"10.5244\/C.28.88"},{"issue":"2","key":"6304_CR28","doi-asserted-by":"publisher","first-page":"239","DOI":"10.1109\/72.80236","volume":"1","author":"ED Karnin","year":"1990","unstructured":"Karnin, E. D. (1990). A simple procedure for pruning back-propagation trained neural networks. IEEE Transactions on Neural Networks, 1(2), 239\u2013242.","journal-title":"IEEE Transactions on Neural Networks"},{"key":"6304_CR29","unstructured":"Kingma, D. P., & Ba, J. (2015). Adam: A method for stochastic optimization. In Proceedings of the ICLR."},{"key":"6304_CR30","unstructured":"LeCun, Y., Denker, J. S., & Solla, S. A. (1989). Optimal brain damage. In Proceedings of the NeurIPS (pp. 598\u2013605)."},{"key":"6304_CR31","unstructured":"Lee, N., Ajanthan, T., & Torr, P. H. S. (2019). SNIP: Single-shot network pruning based on connection sensitivity. In Proceedings of the ICLR."},{"key":"6304_CR32","doi-asserted-by":"crossref","unstructured":"Li, B., Wu, B., Su, J., & Wang, G. (2020). EagleEye: Fast sub-net evaluation for efficient neural network pruning. In Proceedings of the ECCV (pp. 639\u2013654).","DOI":"10.1007\/978-3-030-58536-5_38"},{"key":"6304_CR33","unstructured":"Li, H., Kadav, A., Durdanovic, I., Samet, H., & Graf, H. P. (2017). Pruning filters for efficient convnets. In Proceedings of the ICLR."},{"key":"6304_CR34","doi-asserted-by":"crossref","unstructured":"Lin, S., Ji, R., Yan, C., Zhang, B., Cao, L., Ye, Q., Huang, F., & Doermann, D. (2019). Towards optimal structured cnn pruning via generative adversarial learning. In Proceedings of the CVPR (pp. 2790\u20132799).","DOI":"10.1109\/CVPR.2019.00290"},{"key":"6304_CR35","unstructured":"Liu, J., Xu, Z., Shi, R., Cheung, R. C. C., & So, H. K. (2020). Dynamic sparse training: Find efficient sparse network from scratch with trainable masked layers. In Proceedings of the ICLR."},{"key":"6304_CR36","unstructured":"Louizos, C., Welling, M., & Kingma, D. P. (2018). Learning sparse neural networks through l_0 regularization. In Proceedings of the ICLR."},{"key":"6304_CR37","doi-asserted-by":"crossref","unstructured":"Lu, L., Guo, M., & Renals, S. (2017). Knowledge distillation for small-footprint highway networks. In Proceedings of the ICASSP (pp. 4820\u20134824).","DOI":"10.1109\/ICASSP.2017.7953072"},{"key":"6304_CR38","doi-asserted-by":"crossref","unstructured":"Lym, S., Choukse, E., Zangeneh, S., Wen, W., Sanghavi, S., & Erez, M. (2019). PruneTrain: Fast neural network training by dynamic sparse model reconfiguration. In Proceedings of the SC (pp. 1\u201313).","DOI":"10.1145\/3295500.3356156"},{"issue":"1","key":"6304_CR39","first-page":"1","volume":"18","author":"A Matthews","year":"2017","unstructured":"Matthews, A., van der Wilk, M., Nickson, T., Fujii, K., Boukouvalas, A., Le\u00f3n-Villagr\u00e1, P., Ghahramani, Z., & Hensman, J. (2017). GPflow: A Gaussian process library using tensorflow. JMLR, 18(1), 1\u20136.","journal-title":"JMLR"},{"key":"6304_CR40","unstructured":"Micikevicius, P., Narang, S., Alben, J., Diamos, G. F., Elsen, E., Garc\u00eda, D., Ginsburg, B., Houston, M., Kuchaiev, O., Venkatesh, G., & Wu, H. (2018). Mixed precision training. In Proceedings of the ICLR."},{"issue":"1","key":"6304_CR41","first-page":"1","volume":"9","author":"DC Mocanu","year":"2018","unstructured":"Mocanu, D. C., Mocanu, E., Stone, P., Nguyen, P. H., Gibescu, M., & Liotta, A. (2018). Scalable training of artificial neural networks with adaptive sparse connectivity inspired by network science. Nature, 9(1), 1\u201312.","journal-title":"Nature"},{"key":"6304_CR42","unstructured":"Molchanov, P., Tyree, S., Karras, T., Aila, T., & Kautz, J. (2017). Pruning convolutional neural networks for resource efficient inference. In Proceedings of the ICLR."},{"key":"6304_CR43","unstructured":"Mostafa, H., & Wang, X. (2019). Parameter efficient training of deep convolutional neural networks by dynamic sparse reparameterization. In Proceedings of the ICML (pp. 4646\u20134655)."},{"key":"6304_CR44","unstructured":"Mozer, M., & Smolensky, P. (1988). Skeletonization: A technique for trimming the fat from a network via relevance assessment. In Proc. NeurIPS (pp. 107\u2013115)."},{"issue":"2","key":"6304_CR45","doi-asserted-by":"publisher","first-page":"210","DOI":"10.1109\/TVLSI.2007.912191","volume":"16","author":"S Nadarajah","year":"2008","unstructured":"Nadarajah, S., & Kotz, S. (2008). Exact distribution of the max\/min of two Gaussian random variables. Transactions on VLSI, 16(2), 210\u2013212.","journal-title":"Transactions on VLSI"},{"key":"6304_CR46","unstructured":"Narang, S., Diamos, G., Sengupta, S., & Elsen, E. (2017). Exploring sparsity in recurrent neural networks. In Proceedings of the ICLR."},{"issue":"4","key":"6304_CR47","doi-asserted-by":"publisher","first-page":"473","DOI":"10.1162\/neco.1992.4.4.473","volume":"4","author":"SJ Nowlan","year":"1992","unstructured":"Nowlan, S. J., & Hinton, G. E. (1992). Simplifying neural networks by soft weight-sharing. Neural Computation, 4(4), 473\u2013493.","journal-title":"Neural Computation"},{"key":"6304_CR48","doi-asserted-by":"publisher","first-page":"2163","DOI":"10.1109\/ACCESS.2015.2494536","volume":"3","author":"A Polyak","year":"2015","unstructured":"Polyak, A., & Wolf, L. (2015). Channel-level acceleration of deep face representations. IEEE Access, 3, 2163\u20132175.","journal-title":"IEEE Access"},{"key":"6304_CR49","unstructured":"Simonyan, K., & Zisserman, A. (2015). Very deep convolutional networks for large-scale image recognition. In Proceedings of the ICLR."},{"issue":"1","key":"6304_CR50","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava, N., Hinton, G. E., Krizhevsky, A., Sutskever, I., & Salakhutdinov, R. (2014). Dropout: A simple way to prevent neural networks from overfitting. JMLR, 15(1), 1929\u20131958.","journal-title":"JMLR"},{"key":"6304_CR51","unstructured":"Swersky, K., Snoek, J., & Adams, R. P. (2014). Freeze-thaw Bayesian optimization. arXiv:1406.3896."},{"key":"6304_CR52","unstructured":"Tanaka, H., Kunin, D., Yamins, D. L., & Ganguli, S. (2020). Pruning neural networks without any data by iteratively conserving synaptic flow. In Proceedings of the NeurIPS."},{"key":"6304_CR53","doi-asserted-by":"crossref","unstructured":"Tung, F., & Mori, G. (2019). Similarity-preserving knowledge distillation. In Proceedings of the ICCV (pp. 1365\u20131374).","DOI":"10.1109\/ICCV.2019.00145"},{"key":"6304_CR54","unstructured":"Ullrich, K., Meeds, E., & Welling, M. (2017). Soft weight-sharing for neural network compression. In Proceedings of the ICLR."},{"key":"6304_CR55","unstructured":"Wang, C., Zhang, G., & Grosse, R. B. (2020a). Picking winning tickets before training by preserving gradient flow. In Proceedings of the ICLR."},{"key":"6304_CR56","doi-asserted-by":"crossref","unstructured":"Wang, Y., Zhang, X., Xie, L., Zhou, J., Su, H., Zhang, B., & Hu, X. (2020b). Pruning from scratch. In Proceedings of the AAAI (pp. 12273\u201312280).","DOI":"10.1609\/aaai.v34i07.6910"},{"key":"6304_CR57","unstructured":"Wen, W., Wu, C., Wang, Y., Chen, Y., & Li, H. (2016). Learning structured sparsity in deep neural networks. In Proceedings of the NeurIPS (pp. 2074\u20132082)."},{"key":"6304_CR58","doi-asserted-by":"crossref","unstructured":"Yang, C., Bulu\u00e7, A., & Owens, J. D. (2018). Design principles for sparse matrix multiplication on the GPU. In Proceedings of the Euro-Par (pp. 672\u2013687).","DOI":"10.1007\/978-3-319-96983-1_48"},{"key":"6304_CR59","doi-asserted-by":"crossref","unstructured":"Yim, J., Joo, D., Bae, J., & Kim, J. (2017). A gift from knowledge distillation: Fast optimization, network minimization and transfer learning. In Proc. CVPR (pp. 7130\u20137138).","DOI":"10.1109\/CVPR.2017.754"}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-023-06304-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10994-023-06304-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-023-06304-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,16]],"date-time":"2024-10-16T11:04:35Z","timestamp":1729076675000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10994-023-06304-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,3,14]]},"references-count":59,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2023,7]]}},"alternative-id":["6304"],"URL":"https:\/\/doi.org\/10.1007\/s10994-023-06304-1","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"value":"0885-6125","type":"print"},{"value":"1573-0565","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,3,14]]},"assertion":[{"value":"15 December 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 November 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 January 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 March 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not Applicable.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not Applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}},{"value":"Not Applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not Applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}