{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T19:33:22Z","timestamp":1780774402778,"version":"3.54.1"},"reference-count":43,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,12,24]],"date-time":"2024-12-24T00:00:00Z","timestamp":1734998400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,24]],"date-time":"2024-12-24T00:00:00Z","timestamp":1734998400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,2]]},"DOI":"10.1007\/s10489-024-06207-1","type":"journal-article","created":{"date-parts":[[2024,12,24]],"date-time":"2024-12-24T02:47:35Z","timestamp":1735008455000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Efficient knowledge distillation using a shift window target-aware transformer"],"prefix":"10.1007","volume":"55","author":[{"given":"Jing","family":"Feng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8073-8774","authenticated-orcid":false,"given":"Wen Eng","family":"Ong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,12,24]]},"reference":[{"issue":"12","key":"6207_CR1","doi-asserted-by":"publisher","first-page":"2295","DOI":"10.1109\/JPROC.2017.2761740","volume":"105","author":"V Sze","year":"2017","unstructured":"Sze V, Chen Y-H, Yang T-J, Emer JS (2017) Efficient processing of deep neural networks: a tutorial and survey. Proc IEEE 105(12):2295\u20132329","journal-title":"Proc IEEE"},{"key":"6207_CR2","doi-asserted-by":"publisher","first-page":"1789","DOI":"10.1007\/s11263-021-01453-z","volume":"129","author":"J Gou","year":"2021","unstructured":"Gou J, Yu B, Maybank SJ, Tao D (2021) Knowledge distillation: a survey. Int J Comput Vision 129:1789\u20131819","journal-title":"Int J Comput Vision"},{"key":"6207_CR3","unstructured":"Hinton G, Vinyals O, Dean J (2015) Distilling the knowledge in a neural network. arXiv:1503.02531"},{"key":"6207_CR4","doi-asserted-by":"crossref","unstructured":"Yang C, Yu X, An Z, Xu Y (2023) Categories of response-based, feature-based, and relation-based knowledge distillation. In: Advancements in knowledge distillation: towards new horizons of intelligent systems, Springer, pp 1\u201332","DOI":"10.1007\/978-3-031-32095-8_1"},{"key":"6207_CR5","unstructured":"Yang J, Martinez B, Bulat A, Tzimiropoulos G (2020) Knowledge distillation via adaptive instance normalization. arXiv:2003.04289"},{"key":"6207_CR6","doi-asserted-by":"crossref","unstructured":"Guan Y, Zhao P, Wang B, Zhang Y, Yao C, Bian K, Tang J (2020) Differentiable feature aggregation search for knowledge distillation. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XVII 16, Springer, pp 469\u2013484","DOI":"10.1007\/978-3-030-58520-4_28"},{"key":"6207_CR7","doi-asserted-by":"crossref","unstructured":"Wang X, Fu T, Liao S, Wang S, Lei Z, Mei T (2020) Exclusivity-consistency regularized knowledge distillation for face recognition. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXIV 16, Springer, pp 325\u2013342","DOI":"10.1007\/978-3-030-58586-0_20"},{"key":"6207_CR8","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Advan Neural Inform Process Syst 30"},{"key":"6207_CR9","unstructured":"Zagoruyko S, Komodakis N (2016) Paying more attention to attention: improving the performance of convolutional neural networks via attention transfer. arXiv:1612.03928"},{"key":"6207_CR10","doi-asserted-by":"crossref","unstructured":"Wu Y, Passban P, Rezagholizade M, Liu Q (2020) Why skip if you can combine: a simple knowledge distillation technique for intermediate layers. arXiv:2010.03034","DOI":"10.18653\/v1\/2020.emnlp-main.74"},{"key":"6207_CR11","doi-asserted-by":"crossref","unstructured":"Haidar MA, Anchuri N, Rezagholizadeh M, Ghaddar A, Langlais P, Poupart P (2021) Rail-kd: random intermediate layer mapping for knowledge distillation. arXiv:2109.10164","DOI":"10.18653\/v1\/2022.findings-naacl.103"},{"key":"6207_CR12","doi-asserted-by":"crossref","unstructured":"Wu Y, Rezagholizadeh M, Ghaddar A, Haidar MA, Ghodsi A (2021) Universal-kd: attention-based output-grounded intermediate layer knowledge distillation. In: Proceedings of the 2021 conference on empirical methods in natural language processing, pp 7649\u20137661","DOI":"10.18653\/v1\/2021.emnlp-main.603"},{"key":"6207_CR13","doi-asserted-by":"crossref","unstructured":"Lin S, Xie H, Wang B, Yu K, Chang X, Liang X, Wang G (2022) Knowledge distillation via the target-aware transformer. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10915\u201310924","DOI":"10.1109\/CVPR52688.2022.01064"},{"key":"6207_CR14","unstructured":"Romero A, Ballas N, Kahou SE, Chassang A, Gatta C, Bengio Y (2014) Fitnets: Hints for thin deep nets. arXiv:1412.6550"},{"key":"6207_CR15","unstructured":"Zhang L, Ma K (2020) Improve object detection with feature-based knowledge distillation: towards accurate and efficient detectors. In: International conference on learning representations"},{"key":"6207_CR16","doi-asserted-by":"crossref","unstructured":"Li J, Guo Z, Li H, Han S, Baek J-W, Yang M, Yang R, Suh S (2023) Rethinking feature-based knowledge distillation for face recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 20156\u201320165","DOI":"10.1109\/CVPR52729.2023.01930"},{"key":"6207_CR17","unstructured":"Huang Z, Wang N (2017) Like what you like: knowledge distill via neuron selectivity transfer. arXiv:1707.01219"},{"key":"6207_CR18","doi-asserted-by":"crossref","unstructured":"Heo B, Kim J, Yun S, Park H, Kwak N, Choi JY (2019) A comprehensive overhaul of feature distillation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 1921\u20131930","DOI":"10.1109\/ICCV.2019.00201"},{"key":"6207_CR19","doi-asserted-by":"crossref","unstructured":"Heo B, Lee M, Yun S, Choi JY (2019) Knowledge transfer via distillation of activation boundaries formed by hidden neurons. In: Proceedings of the AAAI conference on artificial intelligence, vol 33, pp 3779\u20133787","DOI":"10.1609\/aaai.v33i01.33013779"},{"key":"6207_CR20","doi-asserted-by":"crossref","unstructured":"Xu K, Rui L, Li Y, Gu L (2020) Feature normalized knowledge distillation for image classification. In: European conference on computer vision, Springer, pp 664\u2013680","DOI":"10.1007\/978-3-030-58595-2_40"},{"key":"6207_CR21","doi-asserted-by":"crossref","unstructured":"Passban P, Wu Y, Rezagholizadeh M, Liu Q (2021) Alp-kd: attention-based layer projection for knowledge distillation. In: Proceedings of the AAAI conference on artificial intelligence, vol 35, pp 13657\u201313665","DOI":"10.1609\/aaai.v35i15.17610"},{"key":"6207_CR22","doi-asserted-by":"crossref","unstructured":"Ji M, Heo B, Park S (2021) Show, attend and distill: Knowledge distillation via attention-based feature matching. In: Proceedings of the AAAI conference on artificial intelligence, vol 35, pp 7945\u20137952","DOI":"10.1609\/aaai.v35i9.16969"},{"key":"6207_CR23","doi-asserted-by":"crossref","unstructured":"Yang C, Zhou H, An Z, Jiang X, Xu Y, Zhang Q (2022) Cross-image relational knowledge distillation for semantic segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12319\u201312328","DOI":"10.1109\/CVPR52688.2022.01200"},{"key":"6207_CR24","doi-asserted-by":"crossref","unstructured":"Yue K, Deng J, Zhou F (2020) Matching guided distillation. In: Computer Vision\u2013ECCV 2020: 16th European Conference,\u00a0Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XV 16, Springer, pp 312\u2013328","DOI":"10.1007\/978-3-030-58555-6_19"},{"key":"6207_CR25","doi-asserted-by":"crossref","unstructured":"Ko J, Park S, Jeong M, Hong S, Ahn E, Chang D-S, Yun S-Y (2023) Revisiting intermediate layer distillation for compressing language models: an overfitting perspective. arXiv:2302.01530","DOI":"10.18653\/v1\/2023.findings-eacl.12"},{"key":"6207_CR26","doi-asserted-by":"crossref","unstructured":"Shu C, Liu Y, Gao J, Yan Z, Shen C (2021) Channel-wise knowledge distillation for dense prediction. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 5311\u20135320","DOI":"10.1109\/ICCV48922.2021.00526"},{"key":"6207_CR27","unstructured":"Krizhevsky A, Hinton G et al (2009) Learning multiple layers of features from tiny images"},{"key":"6207_CR28","doi-asserted-by":"crossref","unstructured":"Deng J, Dong W, Socher R, Li L-J, Li K, Fei-Fei L (2009) Imagenet: a large-scale hierarchical image database. In: 2009 IEEE conference on computer vision and pattern recognition, IEEE, pp. 248\u2013255","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"6207_CR29","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham M, Van Gool L, Williams CK, Winn J, Zisserman A (2010) The pascal visual object classes (voc) challenge. Int J Comput Vision 88:303\u2013338","journal-title":"Int J Comput Vision"},{"key":"6207_CR30","doi-asserted-by":"crossref","unstructured":"Hariharan B, Arbel\u00e1ez P, Bourdev L, Maji S, Malik J (2011) Semantic contours from inverse detectors. In: 2011 International conference on computer vision, IEEE, pp 991\u2013998","DOI":"10.1109\/ICCV.2011.6126343"},{"key":"6207_CR31","doi-asserted-by":"crossref","unstructured":"Caesar H, Uijlings J, Ferrari V (2018) Coco-stuff: Thing and stuff classes in context. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1209\u20131218","DOI":"10.1109\/CVPR.2018.00132"},{"key":"6207_CR32","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL (2014) Microsoft coco: Common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13, Springer, pp 740\u2013755","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"6207_CR33","unstructured":"Loshchilov I, Hutter F (2017) Decoupled weight decay regularization. arXiv:1711.05101"},{"key":"6207_CR34","doi-asserted-by":"crossref","unstructured":"Chen L-C, Zhu Y, Papandreou G, Schroff F, Adam H (2018) Encoder-decoder with atrous separable convolution for semantic image segmentation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 801\u2013818","DOI":"10.1007\/978-3-030-01234-2_49"},{"key":"6207_CR35","unstructured":"Yuan L, Tay FE, Li G, Wang T, Feng J (2019) Revisit knowledge distillation: a teacher-free framework"},{"key":"6207_CR36","doi-asserted-by":"crossref","unstructured":"Park W, Kim D, Lu Y, Cho M (2019) Relational knowledge distillation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3967\u20133976","DOI":"10.1109\/CVPR.2019.00409"},{"key":"6207_CR37","unstructured":"Tian Y, Krishnan D, Isola P (2019) Contrastive representation distillation. arXiv:1910.10699"},{"key":"6207_CR38","doi-asserted-by":"crossref","unstructured":"Tung F, Mori G (2019) Similarity-preserving knowledge distillation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 1365\u20131374","DOI":"10.1109\/ICCV.2019.00145"},{"key":"6207_CR39","doi-asserted-by":"crossref","unstructured":"Peng B, Jin X, Liu J, Li D, Wu Y, Liu Y, Zhou S, Zhang Z (2019) Correlation congruence for knowledge distillation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 5007\u20135016","DOI":"10.1109\/ICCV.2019.00511"},{"key":"6207_CR40","doi-asserted-by":"crossref","unstructured":"Liu L, Huang Q, Lin S, Xie H, Wang B, Chang X, Liang X (2021) Exploring inter-channel correlation for diversity-preserved knowledge distillation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 8271\u20138280","DOI":"10.1109\/ICCV48922.2021.00816"},{"key":"6207_CR41","doi-asserted-by":"publisher","first-page":"302","DOI":"10.1016\/j.neucom.2019.11.118","volume":"406","author":"S Hao","year":"2020","unstructured":"Hao S, Zhou Y, Guo Y (2020) A brief survey on semantic segmentation with deep learning. Neurocomputing 406:302\u2013321","journal-title":"Neurocomputing"},{"key":"6207_CR42","doi-asserted-by":"crossref","unstructured":"Chen P, Liu S, Zhao H, Jia J (2021) Distilling knowledge via knowledge review. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5008\u20135017","DOI":"10.1109\/CVPR46437.2021.00497"},{"key":"6207_CR43","doi-asserted-by":"crossref","unstructured":"Gould S, Fulton R, Koller D (2009) Decomposing a scene into geometric and semantically consistent regions. In: 2009 IEEE 12th international conference on computer vision, IEEE, pp 1\u20138","DOI":"10.1109\/ICCV.2009.5459211"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-06207-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-024-06207-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-06207-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,30]],"date-time":"2025-01-30T16:03:45Z","timestamp":1738253025000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-024-06207-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,24]]},"references-count":43,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2025,2]]}},"alternative-id":["6207"],"URL":"https:\/\/doi.org\/10.1007\/s10489-024-06207-1","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,24]]},"assertion":[{"value":"14 December 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 December 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"All data and samples used in this study comply with ethical standards and have received necessary ethical review board approval. All participants provided informed consent.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical and Informed Consent for Data Used"}},{"value":"The authors declare no potential conflicts of interest or financial support related to this research.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}],"article-number":"215"}}