{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,4]],"date-time":"2025-12-04T09:11:07Z","timestamp":1764839467447,"version":"3.46.0"},"reference-count":73,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"1","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U2336211"],"award-info":[{"award-number":["U2336211"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Major Research Plan of the NSFC","award":["92467206"],"award-info":[{"award-number":["92467206"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2026,1]]},"DOI":"10.1109\/tpami.2025.3607982","type":"journal-article","created":{"date-parts":[[2025,9,9]],"date-time":"2025-09-09T17:31:56Z","timestamp":1757439116000},"page":"542-556","source":"Crossref","is-referenced-by-count":0,"title":["Toward Effective Knowledge Distillation: Navigating Beyond Small-Data Pitfall"],"prefix":"10.1109","volume":"48","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6237-7028","authenticated-orcid":false,"given":"Zhiwei","family":"Hao","sequence":"first","affiliation":[{"name":"School of Information and Electrionics, Beijing Institute of Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianyuan","family":"Guo","sequence":"additional","affiliation":[{"name":"Department of Computer Science, City University of Hong Kong, Hong Kong, SAR, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9761-2702","authenticated-orcid":false,"given":"Kai","family":"Han","sequence":"additional","affiliation":[{"name":"Huawei Noah&#x2019;s Ark Lab, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7532-0496","authenticated-orcid":false,"given":"Han","family":"Hu","sequence":"additional","affiliation":[{"name":"School of Information and Electrionics, Beijing Institute of Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4756-0609","authenticated-orcid":false,"given":"Chang","family":"Xu","sequence":"additional","affiliation":[{"name":"School of Computer Science, Faculty of Engineering, University of Sydney, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2709-4946","authenticated-orcid":false,"given":"Yunhe","family":"Wang","sequence":"additional","affiliation":[{"name":"Huawei Noah&#x2019;s Ark Lab, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02157"},{"key":"ref2","first-page":"1","article-title":"Norm: Knowledge distillation via n-to-one representation matching","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Liu","year":"2023"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58595-2_40"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01165"},{"key":"ref5","first-page":"33716","article-title":"Knowledge distillation from a stronger teacher","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Huang","year":"2022"},{"key":"ref6","first-page":"12084","article-title":"Improved feature distillation via projector ensemble","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chen","year":"2022"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01163"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5963"},{"article-title":"Learning multiple layers of features from tiny images","year":"2009","author":"Krizhevsky","key":"ref9"},{"article-title":"Distilling the knowledge in a neural network","year":"2015","author":"Hinton","key":"ref10"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.4324\/9781410605337-29"},{"key":"ref12","first-page":"562","article-title":"Deeply-supervised nets","volume-title":"Proc. Artif. Intell. Statist.","author":"Lee","year":"2015"},{"key":"ref13","article-title":"VanillaKD: Revisit the power of vanilla knowledge distillation from small scale to large scale","author":"Hao","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"article-title":"Rethinking knowledge distillation via cross-entropy","year":"2022","author":"Yang","key":"ref14"},{"key":"ref15","article-title":"Fitnets: Hints for thin deep nets","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Romero","year":"2015"},{"key":"ref16","first-page":"14759","article-title":"Task-oriented feature distillation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhang","year":"2020"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01174"},{"key":"ref18","first-page":"1","article-title":"Paying more attention to attention: Improving the performance of convolutional neural networks via attention transfer","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Komodakis","year":"2017"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.754"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00511"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00409"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00497"},{"key":"ref23","first-page":"1","article-title":"Function-consistent feature distillation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Liu","year":"2023"},{"key":"ref24","first-page":"1","article-title":"Contrastive representation distillation","volume-title":"Proc. IEEE\/CVF Int. Conf. Learn. Representations","author":"Tian","year":"2020"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20083-0_4"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref27","first-page":"10347","article-title":"Training data-efficient image transformers & distillation through attention","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Touvron","year":"2021"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2010.11929"},{"key":"ref29","first-page":"9164","article-title":"Learning efficient vision transformers via fine-grained manifold distillation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Hao","year":"2022"},{"key":"ref30","first-page":"1","article-title":"Better teacher better student: Dynamic prior knowledge for knowledge distillation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zong","year":"2023"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01064"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1145\/3097983.3098135"},{"key":"ref33","first-page":"1","article-title":"Ensemble distribution distillation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Malinin","year":"2020"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00454"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01103"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00381"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00489"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01065"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref40","first-page":"881","article-title":"MADE: Masked autoencoder for distribution estimation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Germain","year":"2015"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01070"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00010"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3386927"},{"key":"ref44","first-page":"4003","article-title":"CCNet: Extracting high quality monolingual datasets from web crawl data","volume-title":"Proc. Twelfth Lang. Resour. Eval. Conf.","author":"Wenzek","year":"2020"},{"key":"ref45","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.195"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4899-7502-7_79-1"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00020"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00359"},{"article-title":"Resnet strikes back: An improved training procedure in TIMM","year":"2021","author":"Wightman","key":"ref52"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00396"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00526"},{"key":"ref55","first-page":"1","article-title":"Improve object detection with feature-based knowledge distillation: Towards accurate and efficient detectors","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhang","year":"2020"},{"key":"ref56","first-page":"15394","article-title":"PKD: General distillation framework for object detectors via pearson correlation coefficient","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Cao","year":"2022"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00219"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-009-0275-4"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref60","first-page":"91","article-title":"Faster R-CNN: Towards real-time object detection with region proposal networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"28","author":"Ren","year":"2015"},{"key":"ref61","article-title":"Deformable DETR: Deformable transformers for end-to-end object detection","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhu","year":"2021"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00852"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.350"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01228-1_26"},{"key":"ref66","article-title":"Large batch optimization for deep learning: Training bert in 76 minutes","volume-title":"Proc. Int. Conf. Learn. Representations","author":"You","year":"2020"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.308"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00612"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_39"},{"article-title":"BEiT v2: Masked image modeling with vector-quantized visual tokenizers","year":"2022","author":"Peng","key":"ref70"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.01167"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01548"},{"key":"ref73","first-page":"79570","article-title":"One-for-all: Bridge the gap between heterogeneous architectures in knowledge distillation","volume-title":"Proc. 37th Neural Inf. Process. Syst.","author":"Hao","year":"2024"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/11275622\/11154052.pdf?arnumber=11154052","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,4]],"date-time":"2025-12-04T09:07:16Z","timestamp":1764839236000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11154052\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1]]},"references-count":73,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2025.3607982","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"type":"print","value":"0162-8828"},{"type":"electronic","value":"2160-9292"},{"type":"electronic","value":"1939-3539"}],"subject":[],"published":{"date-parts":[[2026,1]]}}}