{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T17:32:54Z","timestamp":1783531974482,"version":"3.55.0"},"publisher-location":"Cham","reference-count":59,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031200823","type":"print"},{"value":"9783031200830","type":"electronic"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-20083-0_8","type":"book-chapter","created":{"date-parts":[[2022,11,2]],"date-time":"2022-11-02T19:46:34Z","timestamp":1667418394000},"page":"120-136","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":31,"title":["Prune Your Model Before Distill It"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9509-3315","authenticated-orcid":false,"given":"Jinhyuk","family":"Park","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6346-4182","authenticated-orcid":false,"given":"Albert","family":"No","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,11,3]]},"reference":[{"issue":"3","key":"8_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3005348","volume":"13","author":"S Anwar","year":"2017","unstructured":"Anwar, S., Hwang, K., Sung, W.: Structured pruning of deep convolutional neural networks. ACM J. Emerg. Technol. Comput. Syst. (JETC) 13(3), 1\u201318 (2017)","journal-title":"ACM J. Emerg. Technol. Comput. Syst. (JETC)"},{"key":"8_CR2","unstructured":"Banner, R., Hubara, I., Hoffer, E., Soudry, D.: Scalable methods for 8-bit training of neural networks. In: NeurIPS (2018)"},{"key":"8_CR3","unstructured":"Chen, T., Kornblith, S., Swersky, K., Norouzi, M., Hinton, G.E.: Big self-supervised models are strong semi-supervised learners. In: NeurIPS (2020)"},{"key":"8_CR4","doi-asserted-by":"crossref","unstructured":"Cho, J.H., Hariharan, B.: On the efficacy of knowledge distillation. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00489"},{"key":"8_CR5","unstructured":"Czarnecki, W.M., Osindero, S., Jaderberg, M., Swirszcz, G., Pascanu, R.: Sobolev training for neural networks. In: NeurIPS (2017)"},{"key":"8_CR6","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: Imagenet: a large-scale hierarchical image database. In: CVPR (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"8_CR7","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. In: NAACL-HLT (2019)"},{"issue":"4","key":"8_CR8","doi-asserted-by":"publisher","first-page":"681","DOI":"10.1007\/s11023-020-09548-1","volume":"30","author":"L Floridi","year":"2020","unstructured":"Floridi, L., Chiriatti, M.: GPT-3: Its Nature, Scope, Limits, and Consequences. Mind. Mach. 30(4), 681\u2013694 (2020). https:\/\/doi.org\/10.1007\/s11023-020-09548-1","journal-title":"Mind. Mach."},{"key":"8_CR9","unstructured":"Frankle, J., Carbin, M.: The lottery ticket hypothesis: finding sparse, trainable neural networks. In: ICLR (2019)"},{"key":"8_CR10","unstructured":"Frankle, J., Dziugaite, G.K., Roy, D., Carbin, M.: Linear mode connectivity and the lottery ticket hypothesis. In: ICML (2020)"},{"key":"8_CR11","unstructured":"Grill, J.B., et al.: Bootstrap your own latent - a new approach to self-supervised learning. In: NeurIPS (2020)"},{"issue":"3","key":"8_CR12","doi-asserted-by":"publisher","first-page":"243","DOI":"10.1145\/3007787.3001163","volume":"44","author":"S Han","year":"2016","unstructured":"Han, S., et al.: Eie: efficient inference engine on compressed deep neural network. ACM SIGARCH Comput. Archit. News 44(3), 243\u2013254 (2016)","journal-title":"ACM SIGARCH Comput. Archit. News"},{"key":"8_CR13","unstructured":"Han, S., Mao, H., Dally, W.J.: Deep compression: compressing deep neural network with pruning, trained quantization and huffman coding. In: ICLR (2016)"},{"key":"8_CR14","unstructured":"Havasi, M., Peharz, R., Hernandez-Lobato, J.M.: Minimal random code learning: getting bits back from compressed model parameters. In: ICLR (2019)"},{"key":"8_CR15","doi-asserted-by":"crossref","unstructured":"He, K., Fan, H., Wu, Y., Xie, S., Girshick, R.: Momentum contrast for unsupervised visual representation learning. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"8_CR16","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"8_CR17","doi-asserted-by":"crossref","unstructured":"He, Y., Kang, G., Dong, X., Fu, Y., Yang, Y.: Soft filter pruning for accelerating deep convolutional neural networks. In: IJCAI (2018)","DOI":"10.24963\/ijcai.2018\/309"},{"key":"8_CR18","doi-asserted-by":"crossref","unstructured":"He, Y., Zhang, X., Sun, J.: Channel pruning for accelerating very deep neural networks. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.155"},{"key":"8_CR19","doi-asserted-by":"crossref","unstructured":"Heo, B., Lee, M., Yun, S., Choi, J.Y.: Knowledge transfer via distillation of activation boundaries formed by hidden neurons. In: AAAI (2019)","DOI":"10.1609\/aaai.v33i01.33013779"},{"key":"8_CR20","unstructured":"Hinton, G., Vinyals, O., Dean, J.: Distilling the knowledge in a neural network. In: NeurIPS Workshop (2015)"},{"key":"8_CR21","unstructured":"Huang, Y., et al.: Gpipe: efficient training of giant neural networks using pipeline parallelism. In: NeurIPS (2019)"},{"key":"8_CR22","doi-asserted-by":"crossref","unstructured":"Idelbayev, Y., Carreira-Perpinan, M.A.: Low-rank compression of neural nets: learning the rank of each layer. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00807"},{"key":"8_CR23","doi-asserted-by":"crossref","unstructured":"Jing, Y., Yang, Y., Wang, X., Song, M., Tao, D.: Amalgamating knowledge from heterogeneous graph neural networks. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.01545"},{"key":"8_CR24","unstructured":"Krizhevsky, A., Hinton, G., et al.: Learning multiple layers of features from tiny images. (Technical report) (2009)"},{"key":"8_CR25","unstructured":"Le, Y., Yang, X.: Tiny imagenet visual recognition challenge. (Technical report) (2015)"},{"key":"8_CR26","unstructured":"LeCun, Y., Denker, J.S., Solla, S.A.: Optimal brain damage. In: NeurIPS (1990)"},{"key":"8_CR27","unstructured":"Lee, J., Park, S., Mo, S., Ahn, S., Shin, J.: Layer-adaptive sparsity for the magnitude-based pruning. In: ICLR (2021)"},{"key":"8_CR28","unstructured":"LeJeune, D., Javadi, H., Baraniuk, R.: The flip side of the reweighted coin: duality of adaptive dropout and regularization. In: NeurIPS (2021)"},{"key":"8_CR29","unstructured":"Li, F., Zhang, B., Liu, B.: Ternary weight networks. arXiv:1605.04711 (2016)"},{"key":"8_CR30","unstructured":"Li, G., et al.: Residual distillation: towards portable deep neural networks without shortcuts. In: NeurIPS (2020)"},{"key":"8_CR31","unstructured":"Li, H., Kadav, A., Durdanovic, I., Samet, H., Graf, H.P.: Pruning filters for efficient convnets. In: ICLR (2017)"},{"key":"8_CR32","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: Metapruning: meta learning for automatic neural network channel pruning. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00339"},{"key":"8_CR33","doi-asserted-by":"crossref","unstructured":"Liu, Z., Li, J., Shen, Z., Huang, G., Yan, S., Zhang, C.: Learning efficient convolutional networks through network slimming. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.298"},{"key":"8_CR34","unstructured":"Liu, Z., Sun, M., Zhou, T., Huang, G., Darrell, T.: Rethinking the value of network pruning. In: ICLR (2018)"},{"key":"8_CR35","doi-asserted-by":"crossref","unstructured":"Luo, J.H., Wu, J., Lin, W.: Thinet: a filter level pruning method for deep neural network compression. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.541"},{"key":"8_CR36","doi-asserted-by":"crossref","unstructured":"Mirzadeh, S.I., Farajtabar, M., Li, A., Levine, N., Matsukawa, A., Ghasemzadeh, H.: Improved knowledge distillation via teacher assistant. In: AAAI (2020)","DOI":"10.1609\/aaai.v34i04.5963"},{"key":"8_CR37","unstructured":"Park, D.Y., Cha, M.H., Jeong, C., Kim, D., Han, B.: Learning student-friendly teacher networks for knowledge distillation. In: NeurIPS (2021)"},{"key":"8_CR38","unstructured":"Patterson, D., et al.: Carbon emissions and large neural network training. arXiv:2104.10350 (2021)"},{"key":"8_CR39","unstructured":"Renda, A., Frankle, J., Carbin, M.: Comparing rewinding and fine-tuning in neural network pruning. In: ICLR (2020)"},{"key":"8_CR40","unstructured":"Romero, A., Ballas, N., Kahou, S.E., Chassang, A., Gatta, C., Bengio, Y.: Fitnets: hints for thin deep nets. In: ICLR (2015)"},{"key":"8_CR41","doi-asserted-by":"crossref","unstructured":"Sainath, T.N., Kingsbury, B., Sindhwani, V., Arisoy, E., Ramabhadran, B.: Low-rank matrix factorization for deep neural network training with high-dimensional output targets. In: ICASSP (2013)","DOI":"10.1109\/ICASSP.2013.6638949"},{"key":"8_CR42","doi-asserted-by":"crossref","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., Chen, L.C.: Mobilenetv 2: inverted residuals and linear bottlenecks. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00474"},{"key":"8_CR43","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. In: ICLR (2015)"},{"key":"8_CR44","unstructured":"Srinivas, S., Fleuret, F.: Knowledge transfer with jacobian matching. In: ICML (2018)"},{"key":"8_CR45","unstructured":"Stanton, S., Izmailov, P., Kirichenko, P., Alemi, A.A., Wilson, A.G.: Does knowledge distillation really work? In: NeurIPS (2021)"},{"key":"8_CR46","doi-asserted-by":"crossref","unstructured":"Su, X., You, S., Wang, F., Qian, C., Zhang, C., Xu, C.: Bcnet: searching for network width with bilaterally coupled network. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00221"},{"key":"8_CR47","unstructured":"Tanaka, H., Kunin, D., Yamins, D.L., Ganguli, S.: Pruning neural networks without any data by iteratively conserving synaptic flow. In: NeurIPS (2020)"},{"key":"8_CR48","unstructured":"Tian, Y., Krishnan, D., Isola, P.: Contrastive representation distillation. In: ICLR (2019)"},{"key":"8_CR49","unstructured":"Wen, W., Wu, C., Wang, Y., Chen, Y., Li, H.: Learning structured sparsity in deep neural networks. In: NeurIPS (2016)"},{"issue":"4","key":"8_CR50","doi-asserted-by":"publisher","first-page":"700","DOI":"10.1109\/JSTSP.2020.2969554","volume":"14","author":"S Wiedemann","year":"2020","unstructured":"Wiedemann, S., et al.: Deepcabac: a universal compression algorithm for deep neural networks. IEEE J. Sel. Top. Sig.l Process. 14(4), 700\u2013714 (2020)","journal-title":"IEEE J. Sel. Top. Sig.l Process."},{"key":"8_CR51","unstructured":"Xu, Z., Hsu, Y.C., Huang, J.: Training shallow and thin networks for acceleration via knowledge distillation with conditional adversarial networks. In: ICLR Workshop (2017)"},{"key":"8_CR52","doi-asserted-by":"crossref","unstructured":"Yang, Y., Qiu, J., Song, M., Tao, D., Wang, X.: Distilling knowledge from graph convolutional networks. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00710"},{"key":"8_CR53","unstructured":"Ye, J., Lu, X., Lin, Z., Wang, J.Z.: Rethinking the smaller-norm-less-informative assumption in channel pruning of convolution layers. In: ICLR (2018)"},{"key":"8_CR54","doi-asserted-by":"crossref","unstructured":"Yuan, L., Tay, F.E., Li, G., Wang, T., Feng, J.: Revisiting knowledge distillation via label smoothing regularization. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00396"},{"key":"8_CR55","doi-asserted-by":"crossref","unstructured":"Zagoruyko, S., Komodakis, N.: Wide residual networks. In: BMVC (2016)","DOI":"10.5244\/C.30.87"},{"key":"8_CR56","unstructured":"Zagoruyko, S., Komodakis, N.: Paying more attention to attention: improving the performance of convolutional neural networks via attention transfer. In: ICLR (2017)"},{"key":"8_CR57","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"662","DOI":"10.1007\/978-3-319-46493-0_40","volume-title":"Computer Vision \u2013 ECCV 2016","author":"H Zhou","year":"2016","unstructured":"Zhou, H., Alvarez, J.M., Porikli, F.: Less is more: towards compact cnns. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9908, pp. 662\u2013677. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46493-0_40"},{"key":"8_CR58","unstructured":"Zhou, H., et al.: Rethinking soft labels for knowledge distillation: a bias\u2013variance tradeoff perspective. In: ICLR (2021)"},{"key":"8_CR59","unstructured":"Zhuang, T., Zhang, Z., Huang, Y., Zeng, X., Shuang, K., Li, X.: Neuron-level structured pruning using polarization regularizer. In: NeurIPS (2020)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-20083-0_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,13]],"date-time":"2024-03-13T11:54:47Z","timestamp":1710330887000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-20083-0_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031200823","9783031200830"],"references-count":59,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-20083-0_8","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"3 November 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tel Aviv","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Israel","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2022.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5804","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1645","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.21","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.91","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}