{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T06:22:13Z","timestamp":1782454933019,"version":"3.54.5"},"reference-count":34,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Korean Government","award":["2021R1A2C3006659"],"award-info":[{"award-number":["2021R1A2C3006659"]}]},{"DOI":"10.13039\/501100010418","name":"Institute for Information and communications Technology Planning and Evaluation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100010418","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Korean Government","award":["RS-2021-II211343"],"award-info":[{"award-number":["RS-2021-II211343"]}]},{"name":"Korean Government","award":["RS-2022-II220320"],"award-info":[{"award-number":["RS-2022-II220320"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2025]]},"DOI":"10.1109\/access.2025.3585106","type":"journal-article","created":{"date-parts":[[2025,7,2]],"date-time":"2025-07-02T13:45:39Z","timestamp":1751463939000},"page":"115548-115557","source":"Crossref","is-referenced-by-count":2,"title":["The Role of Teacher Calibration in Knowledge Distillation"],"prefix":"10.1109","volume":"13","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-1925-6201","authenticated-orcid":false,"given":"Suyoung","family":"Kim","sequence":"first","affiliation":[{"name":"Department of Intelligence and Information, Seoul National University, Gwanak, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Seonguk","family":"Park","sequence":"additional","affiliation":[{"name":"A2Mind, Gangnam, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5891-8081","authenticated-orcid":false,"given":"Junhoo","family":"Lee","sequence":"additional","affiliation":[{"name":"Department of Intelligence and Information, Seoul National University, Gwanak, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1792-0327","authenticated-orcid":false,"given":"Nojun","family":"Kwak","sequence":"additional","affiliation":[{"name":"Department of Intelligence and Information, Seoul National University, Gwanak, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00489"},{"key":"ref2","article-title":"Better teacher better student: Dynamic prior knowledge for knowledge distillation","volume-title":"Proc. 11th Int. Conf. Learn. Represent.","author":"Qiu"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01065"},{"key":"ref4","first-page":"1321","article-title":"On calibration of modern neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Guo"},{"key":"ref5","first-page":"38","article-title":"Measuring calibration in deep learning","volume-title":"Proc. CVPR Workshops","author":"Nixon"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1503.02531"},{"key":"ref7","article-title":"FitNets: Hints for thin deep nets","author":"Romero","year":"2014","journal-title":"arXiv:1412.6550"},{"key":"ref8","article-title":"Paraphrasing complex network: Network compression via factor transfer","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kim"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00454"},{"key":"ref10","first-page":"2006","article-title":"Feature-map-level online adversarial knowledge distillation","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","author":"Chung"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5963"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00926"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02325"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00497"},{"key":"ref15","first-page":"1607","article-title":"Born again neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Furlanello"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01165"},{"key":"ref17","first-page":"21933","article-title":"Respecting transfer gap in knowledge distillation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Niu"},{"key":"ref18","article-title":"Rethinking soft labels for knowledge distillation: A bias-variance tradeoff perspective","author":"Zhou","year":"2021","journal-title":"arXiv:2102.00650"},{"key":"ref19","first-page":"7632","article-title":"A statistical perspective on distillation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Menon"},{"key":"ref20","article-title":"Rethinking the knowledge distillation from the perspective of model calibration","author":"Yang","year":"2021","journal-title":"arXiv:2111.01684"},{"key":"ref21","first-page":"609","article-title":"Obtaining calibrated probability estimates from decision trees and naive Bayesian classifiers","volume-title":"Proc. ICML","author":"Zadrozny"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/775047.775151"},{"key":"ref23","first-page":"61","article-title":"Probabilistic outputs for support vector machines and comparisons to regularized likelihood methods","volume-title":"Advances in Large Margin Classifiers","volume":"10","author":"Platt","year":"1999"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.2307\/2987588"},{"key":"ref25","first-page":"1","article-title":"The implicit bias of gradient descent on separable data","volume":"19","author":"Soudry","year":"2018","journal-title":"J. Mach. Learn. Res."},{"key":"ref26","volume-title":"Detectron2","author":"Wu","year":"2019"},{"key":"ref27","article-title":"Contrastive representation distillation","author":"Tian","year":"2019","journal-title":"arXiv:1910.10699"},{"key":"ref28","article-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky","year":"2009"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref31","first-page":"91","article-title":"Faster R-CNN: Towards real-time object detection with region proposal networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"28","author":"Ren"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.48550\/arxiv.1710.09412"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.2172\/1525811"},{"key":"ref34","first-page":"26135","article-title":"When and how mixup improves calibration","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zhang"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6287639\/10820123\/11062864.pdf?arnumber=11062864","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,11]],"date-time":"2025-07-11T17:46:06Z","timestamp":1752255966000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11062864\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":34,"URL":"https:\/\/doi.org\/10.1109\/access.2025.3585106","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]}}}