{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T18:00:24Z","timestamp":1772906424150,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":39,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,24]],"date-time":"2024-08-24T00:00:00Z","timestamp":1724457600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,25]]},"DOI":"10.1145\/3637528.3671851","type":"proceedings-article","created":{"date-parts":[[2024,8,25]],"date-time":"2024-08-25T04:55:12Z","timestamp":1724561712000},"page":"4278-4289","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Knowledge Distillation with Perturbed Loss: From a Vanilla Teacher to a Proxy Teacher"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7136-7913","authenticated-orcid":false,"given":"Rongzhi","family":"Zhang","sequence":"first","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, GA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0467-4956","authenticated-orcid":false,"given":"Jiaming","family":"Shen","sequence":"additional","affiliation":[{"name":"Google, New York City, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4497-3317","authenticated-orcid":false,"given":"Tianqi","family":"Liu","sequence":"additional","affiliation":[{"name":"Google, New York City, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8721-8656","authenticated-orcid":false,"given":"Jialu","family":"Liu","sequence":"additional","affiliation":[{"name":"Google, New York City, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2941-6240","authenticated-orcid":false,"given":"Michael","family":"Bendersky","sequence":"additional","affiliation":[{"name":"Google, Mountain View, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1423-0854","authenticated-orcid":false,"given":"Marc","family":"Najork","sequence":"additional","affiliation":[{"name":"Google, Mountain View, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3009-598X","authenticated-orcid":false,"given":"Chao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, GA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,8,24]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"GKD: Generalized Knowledge Distillation for Auto-regressive Sequence Models. arXiv preprint arXiv:2306.13649","author":"Agarwal Rishabh","year":"2023","unstructured":"Rishabh Agarwal, Nino Vieillard, Piotr Stanczyk, Sabela Ramos, Matthieu Geist, and Olivier Bachem. 2023. GKD: Generalized Knowledge Distillation for Auto-regressive Sequence Models. arXiv preprint arXiv:2306.13649 (2023)."},{"key":"e_1_3_2_2_2_1","unstructured":"Zeyuan Allen-Zhu and Yuanzhi Li. 2023. Towards Understanding Ensemble Knowledge Distillation and Self-Distillation in Deep Learning. In ICLR."},{"key":"e_1_3_2_2_3_1","volume-title":"Theory of classification: A survey of some recent advances. ESAIM: probability and statistics","author":"Boucheron St\u00e9phane","year":"2005","unstructured":"St\u00e9phane Boucheron, Olivier Bousquet, and G\u00e1bor Lugosi. 2005. Theory of classification: A survey of some recent advances. ESAIM: probability and statistics, Vol. 9 (2005), 323--375."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/1150402.1150464"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00497"},{"key":"e_1_3_2_2_6_1","volume-title":"BoolQ: Exploring the surprising difficulty of natural yes\/no questions. arXiv preprint arXiv:1905.10044","author":"Clark Christopher","year":"2019","unstructured":"Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, and Kristina Toutanova. 2019. BoolQ: Exploring the surprising difficulty of natural yes\/no questions. arXiv preprint arXiv:1905.10044 (2019)."},{"key":"e_1_3_2_2_7_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_2_8_1","volume-title":"Proceedings of the Third International Workshop on Paraphrasing (IWP2005)","author":"William","unstructured":"William B. Dolan and Chris Brockett. 2005. Automatically Constructing a Corpus of Sentential Paraphrases. In Proceedings of the Third International Workshop on Paraphrasing (IWP2005). https:\/\/aclanthology.org\/I05--5002"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3449639.3459277"},{"key":"e_1_3_2_2_10_1","volume-title":"Knowledge Distillation of Large Language Models. arXiv preprint arXiv:2306.08543","author":"Gu Yuxian","year":"2023","unstructured":"Yuxian Gu, Li Dong, Furu Wei, and Minlie Huang. 2023. Knowledge Distillation of Large Language Models. arXiv preprint arXiv:2306.08543 (2023)."},{"key":"e_1_3_2_2_11_1","unstructured":"Geoffrey Hinton Oriol Vinyals Jeff Dean et al. 2015. Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531 Vol. 2 7 (2015)."},{"key":"e_1_3_2_2_12_1","volume-title":"Generalization bounds via distillation. arXiv preprint arXiv:2104.05641","author":"Hsu Daniel","year":"2021","unstructured":"Daniel Hsu, Ziwei Ji, Matus Telgarsky, and Lan Wang. 2021. Generalization bounds via distillation. arXiv preprint arXiv:2104.05641 (2021)."},{"key":"e_1_3_2_2_13_1","volume-title":"Do we need zero training loss after achieving zero training error? arXiv preprint arXiv:2002.08709","author":"Ishida Takashi","year":"2020","unstructured":"Takashi Ishida, Ikko Yamane, Tomoya Sakai, Gang Niu, and Masashi Sugiyama. 2020. Do we need zero training loss after achieving zero training error? arXiv preprint arXiv:2002.08709 (2020)."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.eacl-main.212"},{"key":"e_1_3_2_2_15_1","first-page":"20823","article-title":"Knowledge distillation in wide neural networks: Risk bound, data efficiency and imperfect teacher","volume":"33","author":"Ji Guangda","year":"2020","unstructured":"Guangda Ji and Zhanxing Zhu. 2020. Knowledge distillation in wide neural networks: Risk bound, data efficiency and imperfect teacher. Advances in Neural Information Processing Systems, Vol. 33 (2020), 20823--20833.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_16_1","volume-title":"Sequence-level knowledge distillation. arXiv preprint arXiv:1606.07947","author":"Kim Yoon","year":"2016","unstructured":"Yoon Kim and Alexander M Rush. 2016. Sequence-level knowledge distillation. arXiv preprint arXiv:1606.07947 (2016)."},{"key":"e_1_3_2_2_17_1","unstructured":"Solomon Kullback. 1959. Statistics and information theory."},{"key":"e_1_3_2_2_18_1","volume-title":"PolyLoss: A Polynomial Expansion Perspective of Classification Loss Functions. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=gSdSJoenupI","author":"Leng Zhaoqi","year":"2022","unstructured":"Zhaoqi Leng, Mingxing Tan, Chenxi Liu, Ekin Dogus Cubuk, Jay Shi, Shuyang Cheng, and Dragomir Anguelov. 2022. PolyLoss: A Polynomial Expansion Perspective of Classification Loss Functions. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=gSdSJoenupI"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.324"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00171"},{"key":"e_1_3_2_2_21_1","volume-title":"Aditya Krishna Menon, and Sanjiv Kumar","author":"Lukasik Michal","year":"2021","unstructured":"Michal Lukasik, Srinadh Bhojanapalli, Aditya Krishna Menon, and Sanjiv Kumar. 2021. Teacher's pet: understanding and mitigating biases in distillation. arXiv preprint arXiv:2106.10494 (2021)."},{"key":"e_1_3_2_2_22_1","volume-title":"International Conference on Machine Learning. PMLR, 7632--7642","author":"Menon Aditya K","year":"2021","unstructured":"Aditya K Menon, Ankit Singh Rawat, Sashank Reddi, Seungyeon Kim, and Sanjiv Kumar. 2021. A statistical perspective on distillation. In International Conference on Machine Learning. PMLR, 7632--7642."},{"key":"e_1_3_2_2_23_1","volume-title":"Hinton","author":"M\u00fcller Rafael","year":"2019","unstructured":"Rafael M\u00fcller, Simon Kornblith, and Geoffrey E. Hinton. 2019. When Does Label Smoothing Help?. In NeurIPS."},{"key":"e_1_3_2_2_24_1","unstructured":"OpenAI. 2022. ChatGPT."},{"key":"e_1_3_2_2_26_1","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, Peter J Liu, et al. 2020. Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res., Vol. 21, 140 (2020), 1--67.","journal-title":"J. Mach. Learn. Res."},{"key":"e_1_3_2_2_27_1","volume-title":"Better Supervisory Signals by Observing Learning Paths. arXiv preprint arXiv:2203.02485","author":"Ren Yi","year":"2022","unstructured":"Yi Ren, Shangmin Guo, and Danica J Sutherland. 2022. Better Supervisory Signals by Observing Learning Paths. arXiv preprint arXiv:2203.02485 (2022)."},{"key":"e_1_3_2_2_28_1","first-page":"6906","article-title":"Does knowledge distillation really work","volume":"34","author":"Stanton Samuel","year":"2021","unstructured":"Samuel Stanton, Pavel Izmailov, Polina Kirichenko, Alexander A Alemi, and Andrew G Wilson. 2021. Does knowledge distillation really work? Advances in Neural Information Processing Systems, Vol. 34 (2021), 6906--6919.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.308"},{"key":"e_1_3_2_2_30_1","volume-title":"Contrastive representation distillation. arXiv preprint arXiv:1910.10699","author":"Tian Yonglong","year":"2019","unstructured":"Yonglong Tian, Dilip Krishnan, and Phillip Isola. 2019. Contrastive representation distillation. arXiv preprint arXiv:1910.10699 (2019)."},{"key":"e_1_3_2_2_31_1","volume-title":"GLUE: A multi-task benchmark and analysis platform for natural language understanding. arXiv preprint arXiv:1804.07461","author":"Wang Alex","year":"2018","unstructured":"Alex Wang, Amanpreet Singh, Julian Michael, Felix Hill, Omer Levy, and Samuel R Bowman. 2018. GLUE: A multi-task benchmark and analysis platform for natural language understanding. arXiv preprint arXiv:1804.07461 (2018)."},{"key":"e_1_3_2_2_32_1","volume-title":"Neural Network Acceptability Judgments. arXiv preprint arXiv:1805.12471","author":"Warstadt Alex","year":"2018","unstructured":"Alex Warstadt, Amanpreet Singh, and Samuel R Bowman. 2018. Neural Network Acceptability Judgments. arXiv preprint arXiv:1805.12471 (2018)."},{"key":"e_1_3_2_2_33_1","volume-title":"f-Divergence Minimization for Sequence-Level Knowledge Distillation. arXiv preprint arXiv:2307.15190","author":"Wen Yuqiao","year":"2023","unstructured":"Yuqiao Wen, Zichao Li, Wenyu Du, and Lili Mou. 2023. f-Divergence Minimization for Sequence-Level Knowledge Distillation. arXiv preprint arXiv:2307.15190 (2023)."},{"key":"e_1_3_2_2_34_1","volume-title":"Examining the Inductive Bias of Neural Language Models with Artificial Languages. In Annual Meeting of the Association for Computational Linguistics. https:\/\/api.semanticscholar.org\/CorpusID:235293810","author":"Jennifer","unstructured":"Jennifer C. White and Ryan Cotterell. 2021. Examining the Inductive Bias of Neural Language Models with Artificial Languages. In Annual Meeting of the Association for Computational Linguistics. https:\/\/api.semanticscholar.org\/CorpusID:235293810"},{"key":"e_1_3_2_2_35_1","volume-title":"A broad-coverage challenge corpus for sentence understanding through inference. arXiv preprint arXiv:1704.05426","author":"Williams Adina","year":"2017","unstructured":"Adina Williams, Nikita Nangia, and Samuel R Bowman. 2017. A broad-coverage challenge corpus for sentence understanding through inference. arXiv preprint arXiv:1704.05426 (2017)."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00396"},{"key":"e_1_3_2_2_37_1","volume-title":"PLaD: Preference-based Large Language Model Distillation with Pseudo-Preference Pairs. arXiv preprint arXiv:2406.02886","author":"Zhang Rongzhi","year":"2024","unstructured":"Rongzhi Zhang, Jiaming Shen, Tianqi Liu, Haorui Wang, Zhen Qin, Feng Han, Jialu Liu, Simon Baumgartner, Michael Bendersky, and Chao Zhang. 2024. PLaD: Preference-based Large Language Model Distillation with Pseudo-Preference Pairs. arXiv preprint arXiv:2406.02886 (2024)."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01165"},{"key":"e_1_3_2_2_39_1","volume-title":"Rethinking soft labels for knowledge distillation: A bias-variance tradeoff perspective. arXiv preprint arXiv:2102.00650","author":"Zhou Helong","year":"2021","unstructured":"Helong Zhou, Liangchen Song, Jiajie Chen, Ye Zhou, Guoli Wang, Junsong Yuan, and Qian Zhang. 2021. Rethinking soft labels for knowledge distillation: A bias-variance tradeoff perspective. arXiv preprint arXiv:2102.00650 (2021)."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.485"}],"event":{"name":"KDD '24: The 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Barcelona Spain","acronym":"KDD '24","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3637528.3671851","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3637528.3671851","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:04:14Z","timestamp":1750291454000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3637528.3671851"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,24]]},"references-count":39,"alternative-id":["10.1145\/3637528.3671851","10.1145\/3637528"],"URL":"https:\/\/doi.org\/10.1145\/3637528.3671851","relation":{},"subject":[],"published":{"date-parts":[[2024,8,24]]},"assertion":[{"value":"2024-08-24","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}