{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,4]],"date-time":"2026-04-04T17:59:38Z","timestamp":1775325578247,"version":"3.50.1"},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2025,7,21]],"date-time":"2025-07-21T00:00:00Z","timestamp":1753056000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,21]],"date-time":"2025-07-21T00:00:00Z","timestamp":1753056000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-025-07661-5","type":"journal-article","created":{"date-parts":[[2025,7,21]],"date-time":"2025-07-21T18:14:29Z","timestamp":1753121669000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["A multilevel integrated supervision self-distillation method"],"prefix":"10.1007","volume":"81","author":[{"given":"Yuxiang","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuemei","family":"Lei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,7,21]]},"reference":[{"key":"7661_CR1","doi-asserted-by":"publisher","unstructured":"Liu W, Anguelov D, Erhan D, Szegedy C, Reed S, Fu CY, Berg AC (2016) SSD: single shot multibox detector. In: Computer vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part I 14. Springer, pp 21\u201337. https:\/\/doi.org\/10.1007\/978-3-319-46448-0_2","DOI":"10.1007\/978-3-319-46448-0_2"},{"issue":"1s","key":"7661_CR2","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3419842","volume":"17","author":"J Xiao","year":"2021","unstructured":"Xiao J, Xu H, Gao H, Bian M, Li Y (2021) A weakly supervised semantic segmentation network by aggregating seed cues: the multi-object proposal generation perspective. ACM Trans Multimedia Comput Commun Appl 17(1s):1\u201319. https:\/\/doi.org\/10.1145\/3419842","journal-title":"ACM Trans Multimedia Comput Commun Appl"},{"issue":"6","key":"7661_CR3","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1145\/3065386","volume":"60","author":"A Krizhevsky","year":"2017","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2017) Imagenet classification with deep convolutional neural networks. Commun ACM 60(6):84\u201390. https:\/\/doi.org\/10.1145\/3065386","journal-title":"Commun ACM"},{"key":"7661_CR4","doi-asserted-by":"publisher","unstructured":"Zhang L, Song J, Gao A, Chen J, Bao C, Ma K (2019) Be your own teacher: improve the performance of convolutional neural networks via self distillation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp 3713\u20133722. https:\/\/doi.org\/10.48550\/arXiv.1905.08094","DOI":"10.48550\/arXiv.1905.08094"},{"key":"7661_CR5","doi-asserted-by":"publisher","unstructured":"Liu Z, Mao H, Wu CY, Feichtenhofer C, Darrell T, Xie S (2022) A convnet for the 2020s. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 11976\u201311986.https:\/\/doi.org\/10.48550\/arXiv.2201.03545","DOI":"10.48550\/arXiv.2201.03545"},{"key":"7661_CR6","doi-asserted-by":"publisher","unstructured":"Howard AG (2017) Mobilenets: efficient convolutional neural networks for mobilevision applications. arXiv preprint arXiv:1704.04861. https:\/\/doi.org\/10.48550\/arXiv.1704.04861","DOI":"10.48550\/arXiv.1704.04861"},{"key":"7661_CR7","doi-asserted-by":"publisher","unstructured":"Courbariaux M, Bengio Y, David JP (2015) BinaryConnect: training deep neural networks with binary weights during propagations. arXiv preprint arXiv:1511.00363. https:\/\/doi.org\/10.48550\/arXiv.1511.00363","DOI":"10.48550\/arXiv.1511.00363"},{"key":"7661_CR8","doi-asserted-by":"publisher","unstructured":"Kim E, Ahn C, Oh S (2018) Nestednet: learning nested sparse structures in deep neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 8669\u20138678. https:\/\/doi.org\/10.1109\/CVPR.2018.00904","DOI":"10.1109\/CVPR.2018.00904"},{"key":"7661_CR9","doi-asserted-by":"publisher","unstructured":"Li Y, Chen Y, Dai X, Chen D, Liu M, Yuan L et al (2020) MicroNet: towards image recognition with extremely low FLOPs. arXiv preprint arXiv:2011.12289. https:\/\/doi.org\/10.48550\/arXiv.2011.12289","DOI":"10.48550\/arXiv.2011.12289"},{"key":"7661_CR10","doi-asserted-by":"publisher","unstructured":"Hinton G (2015) Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531. https:\/\/doi.org\/10.48550\/arXiv.1503.02531","DOI":"10.48550\/arXiv.1503.02531"},{"key":"7661_CR11","doi-asserted-by":"publisher","unstructured":"Romero A, Ballas N, Kahou SE, Chassang A, Gatta C, Bengio Y (2014) Fitnets: HINTS for thin deep nets. arXiv preprint arXiv:1412.6550. https:\/\/doi.org\/10.48550\/arXiv.1412.6550","DOI":"10.48550\/arXiv.1412.6550"},{"key":"7661_CR12","doi-asserted-by":"publisher","unstructured":"Zagoruyko S, Komodakis N (2016) Paying more attention to attention: improving the performance of convolutional neural networks via attention transfer. arXiv preprint arXiv:1612.03928. https:\/\/doi.org\/10.48550\/arXiv.1612.03928","DOI":"10.48550\/arXiv.1612.03928"},{"key":"7661_CR13","doi-asserted-by":"publisher","unstructured":"Park W, Kim D, Lu Y, Cho M (2019) Relational knowledge distillation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 3967\u20133976. https:\/\/doi.org\/10.1109\/CVPR.2019.00409","DOI":"10.1109\/CVPR.2019.00409"},{"issue":"6","key":"7661_CR14","doi-asserted-by":"publisher","first-page":"3048","DOI":"10.1109\/TPAMI.2021.3055564","volume":"44","author":"L Wang","year":"2021","unstructured":"Wang L, Yoon KJ (2021) Knowledge distillation and student-teacher learning for visual intelligence: a review and new outlooks. IEEE Trans Pattern Anal Mach Intell 44(6):3048\u20133068. https:\/\/doi.org\/10.1109\/TPAMI.2021.3055564","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"7661_CR15","doi-asserted-by":"publisher","first-page":"12","DOI":"10.1016\/j.inffus.2022.09.007","volume":"90","author":"Z Long","year":"2023","unstructured":"Long Z, Ma F, Sun B, Tan M, Li S (2023) Diversified branch fusion for self-knowledge distillation. Inf Fusion 90:12\u201322. https:\/\/doi.org\/10.1016\/j.inffus.2022.09.007","journal-title":"Inf Fusion"},{"key":"7661_CR16","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2023.119859","volume":"654","author":"P Liang","year":"2024","unstructured":"Liang P, Zhang W, Wang J, Guo Y (2024) Neighbor self-knowledge distillation. Inf Sci 654:119859. https:\/\/doi.org\/10.1016\/j.ins.2023.119859","journal-title":"Inf Sci"},{"key":"7661_CR17","doi-asserted-by":"publisher","first-page":"195","DOI":"10.1109\/OJCS.2023.3288227","volume":"4","author":"S Ni","year":"2023","unstructured":"Ni S, Ma X, Zhu M, Li X, Zhang YD (2023) Reverse self-distillation overcoming the self-distillation barrier. IEEE Open J Comput Soc 4:195\u2013205. https:\/\/doi.org\/10.1109\/OJCS.2023.3288227","journal-title":"IEEE Open J Comput Soc"},{"key":"7661_CR18","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2024.112751","volume":"306","author":"M Zhou","year":"2024","unstructured":"Zhou M, Su Z, Li M, Wang Y, Li G (2024) CSDD-net: a cross semi-supervised dual-feature distillation network for industrial defect detection. Knowl-Based Syst 306:112751. https:\/\/doi.org\/10.1016\/j.knosys.2024.112751","journal-title":"Knowl-Based Syst"},{"key":"7661_CR19","doi-asserted-by":"publisher","DOI":"10.1016\/j.aei.2024.102611","volume":"62","author":"Z Su","year":"2024","unstructured":"Su Z, Zhou M, Li M, Zhang Z, Han D, Li G (2024) Revisiting the application of twin connected parallel networks and regression loss functions in industrial defect detection. Adv Eng Inform 62:102611. https:\/\/doi.org\/10.1016\/j.aei.2024.102611","journal-title":"Adv Eng Inform"},{"issue":"9","key":"7661_CR20","doi-asserted-by":"publisher","first-page":"4769","DOI":"10.1109\/TCSVT.2023.3250031","volume":"33","author":"A L\u00f3pez-Cifuentes","year":"2023","unstructured":"L\u00f3pez-Cifuentes A, Escudero-Vi\u00f1olo M, Besc\u00f3s J, San Miguel JC (2023) Attention-based knowledge distillation in scene recognition: the impact of a DCT-driven loss. IEEE Trans Circuits Syst Video Technol 33(9):4769\u20134783. https:\/\/doi.org\/10.1109\/TCSVT.2023.3250031","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"7661_CR21","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2024.3476068","author":"S Li","year":"2024","unstructured":"Li S, Su T, Zhang X, Wang Z (2024) Continual learning with knowledge distillation: a survey. IEEE Trans Neural Netw Learn Syst. https:\/\/doi.org\/10.1109\/TNNLS.2024.3476068","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"7661_CR22","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2024.3428519","author":"K Acharya","year":"2024","unstructured":"Acharya K, Velasquez A, Song HH (2024) A survey on symbolic knowledge distillation of large language models. IEEE Trans Artif Intell. https:\/\/doi.org\/10.1109\/TAI.2024.3428519","journal-title":"IEEE Trans Artif Intell"},{"issue":"8","key":"7661_CR23","doi-asserted-by":"publisher","first-page":"4388","DOI":"10.1109\/TPAMI.2021.3067100","volume":"44","author":"L Zhang","year":"2021","unstructured":"Zhang L, Bao C, Ma K (2021) Self-distillation: towards efficient and compact neural networks. IEEE Trans Pattern Anal Mach Intell 44(8):4388\u20134403. https:\/\/doi.org\/10.1109\/TPAMI.2021.3067100","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"7661_CR24","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2024.111762","volume":"161","author":"X Zhang","year":"2024","unstructured":"Zhang X, Zhu J, Wang D, Wang Y, Liang T, Wang H, Yin Y (2024) A gradual self distillation network with adaptive channel attention for facial expression recognition. Appl Soft Comput 161:111762. https:\/\/doi.org\/10.1016\/j.asoc.2024.111762","journal-title":"Appl Soft Comput"},{"key":"7661_CR25","doi-asserted-by":"publisher","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 770\u2013778. https:\/\/doi.org\/10.48550\/arXiv.1512.03385","DOI":"10.48550\/arXiv.1512.03385"},{"key":"7661_CR26","doi-asserted-by":"publisher","unstructured":"Wang Q, Wu B, Zhu P, Li P, Zuo W, Hu Q (2020) ECA-Net: efficient channel attention for deep convolutional neural networks. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), June 2020 IEEE, pp 11531\u201311539. https:\/\/doi.org\/10.1109\/CVPR42600.2020.01155","DOI":"10.1109\/CVPR42600.2020.01155"},{"issue":"3","key":"7661_CR27","doi-asserted-by":"publisher","first-page":"1010","DOI":"10.1109\/TITS.2018.2838132","volume":"20","author":"X Hu","year":"2019","unstructured":"Hu X, Xu X, Xiao Y et al (2019) Sinet: a scale-insensitive convolutional neural network for fast vehicle detection. IEEE Trans Intell Transp Syst 20(3):1010\u20131019. https:\/\/doi.org\/10.1109\/TITS.2018.2838132","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"7661_CR28","doi-asserted-by":"publisher","unstructured":"Sandler M, Howard A, Zhu M, Zhmoginov A, Chen LC (2018) Mobilenetv2: inverted residuals and linear bottlenecks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 4510\u20134520. https:\/\/doi.org\/10.1109\/CVPR.2018.00474","DOI":"10.1109\/CVPR.2018.00474"},{"key":"7661_CR29","doi-asserted-by":"publisher","unstructured":"Howard A, Sandler M, Chu G, Chen LC, Chen B, Tan M et al (2019) Searching for MobileNetV3. arXiv preprint arXiv:1905.02244. https:\/\/doi.org\/10.48550\/arXiv.1905.02244","DOI":"10.48550\/arXiv.1905.02244"},{"key":"7661_CR30","doi-asserted-by":"publisher","unstructured":"Tan M, Le QV (2021) EfficientNetV2: smaller models and faster training. arXiv preprint arXiv:2104.00298. https:\/\/doi.org\/10.48550\/arXiv.2104.00298","DOI":"10.48550\/arXiv.2104.00298"},{"key":"7661_CR31","unstructured":"Krizhevsky A, Hinton G (2009) Learning multiple layers of features from tiny images. In: Handbook Systemic Autoimmune Diseases 1(4)"},{"key":"7661_CR32","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky O et al (2015) Imagenet large scale visual recognition challenge. Int J Comput Vis 115:211\u2013252","journal-title":"Int J Comput Vis"},{"key":"7661_CR33","unstructured":"Wah C, Branson S, Welinder P, Perona P, Belongie S (2011) The caltech-UCSD birds-200\u20132011 dataset. California Institute of Technology, Pasadena, CA, USA"},{"key":"7661_CR34","doi-asserted-by":"crossref","unstructured":"Quattoni A, Torralba A (2009) Recognizing indoor scenes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 413\u2013420","DOI":"10.1109\/CVPR.2009.5206537"},{"key":"7661_CR35","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2024.111692","volume":"293","author":"H Kim","year":"2024","unstructured":"Kim H, Suh S, Baek S, Kim D, Jeong D, Cho H, Kim J (2024) AI-KD: adversarial learning and implicit regularization for self-knowledge distillation. Knowl-Based Syst 293:111692. https:\/\/doi.org\/10.1016\/j.knosys.2024.111692","journal-title":"Knowl-Based Syst"},{"key":"7661_CR36","doi-asserted-by":"publisher","unstructured":"Woo S, Park J, Lee JY, Kweon IS (2018) Cbam: convolutional block attention module. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 3\u201319. https:\/\/doi.org\/10.48550\/arXiv.1807.06521.","DOI":"10.48550\/arXiv.1807.06521"},{"key":"7661_CR37","doi-asserted-by":"publisher","unstructured":"Misra D, Nalamada T, Arasanipalai AU, Hou Q (2021) Rotate to attend: convolutional triplet attention module. In: 2021 IEEE Winter Conference on Applications of Computer Vision (WACV), 2021, January. IEEE Computer Society, pp 3138\u20133147. https:\/\/doi.org\/10.1109\/WACV48630.2021.00318.","DOI":"10.1109\/WACV48630.2021.00318"},{"key":"7661_CR38","doi-asserted-by":"publisher","unstructured":"Hou Q, Zhou D, Feng J (2021) Coordinate attention for efficient mobile network design. arXiv preprint arXiv:2103.02907. https:\/\/doi.org\/10.48550\/arXiv.2103.02907","DOI":"10.48550\/arXiv.2103.02907"},{"key":"7661_CR39","doi-asserted-by":"crossref","unstructured":"Han D, Ye T, Han Y, Xia Z, Pan S, Wan P, Song S, Huang G (2024) Agent attention: on the integration of softmax and linear attention. In: European Conference on Computer Vision. Springer, Cham, 2024, September, pp 124\u2013140. https:\/\/arxiv.org\/abs\/2312.08874.","DOI":"10.1007\/978-3-031-72973-7_8"},{"issue":"86","key":"7661_CR40","first-page":"2579","volume":"9","author":"L van der Maaten","year":"2008","unstructured":"van der Maaten L, Hinton G (2008) Visualizing data using t-sne. J Mach Learn Res 9(86):2579\u20132605","journal-title":"J Mach Learn Res"},{"key":"7661_CR41","doi-asserted-by":"publisher","unstructured":"McInnes L, Healy J, Melville J (2018) Umap: uniform manifold approximation and projection for dimension reduction. arXiv preprint arXiv:1802.03426. https:\/\/doi.org\/10.48550\/arXiv.1802.03426","DOI":"10.48550\/arXiv.1802.03426"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-07661-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-025-07661-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-07661-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,7]],"date-time":"2025-09-07T16:40:07Z","timestamp":1757263207000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-025-07661-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,21]]},"references-count":41,"journal-issue":{"issue":"11","published-online":{"date-parts":[[2025,7]]}},"alternative-id":["7661"],"URL":"https:\/\/doi.org\/10.1007\/s11227-025-07661-5","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,21]]},"assertion":[{"value":"3 July 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 July 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}],"article-number":"1178"}}