{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T13:12:43Z","timestamp":1778764363904,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":55,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"PCL","award":["PCL2023A08"],"award-info":[{"award-number":["PCL2023A08"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681462","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:33Z","timestamp":1729925973000},"page":"10487-10496","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["CoTuning: A Large-Small Model Collaborating Distillation Framework for Better Model Generalization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-4014-1543","authenticated-orcid":false,"given":"Zimo","family":"Liu","sequence":"first","affiliation":[{"name":"Peng Cheng Laboratory, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7054-3981","authenticated-orcid":false,"given":"Kangjun","family":"Liu","sequence":"additional","affiliation":[{"name":"Peng Cheng Laboratory, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2348-7530","authenticated-orcid":false,"given":"Mingyue","family":"Guo","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9053-9314","authenticated-orcid":false,"given":"Shiliang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2197-9038","authenticated-orcid":false,"given":"Yaowei","family":"Wang","sequence":"additional","affiliation":[{"name":"Peng Cheng Laboratory &amp; Harbin Institute of Technology, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jksuci.2023.101616"},{"key":"e_1_3_2_1_2_1","volume-title":"ML-LMCL: Mutual Learning and Large-Margin Contrastive Learning for Improving ASR Robustness in Spoken Language Understanding","author":"Cheng Xuxin","unstructured":"Xuxin Cheng, Bowen Cao, Qichen Ye, Zhihong Zhu, Hongxiang Li, and Yuexian Zou. 2023. ML-LMCL: Mutual Learning and Large-Margin Contrastive Learning for Improving ASR Robustness in Spoken Language Understanding. In Association for Computational Linguistics. 6492--6505."},{"key":"e_1_3_2_1_3_1","volume-title":"Mixture-of-Domain-Adapters: Decoupling and Injecting Domain Knowledge to Pre-trained Language Models' Memories","author":"Diao Shizhe","unstructured":"Shizhe Diao, Tianyang Xu, Ruijia Xu, Jiawei Wang, and Tong Zhang. 2023. Mixture-of-Domain-Adapters: Decoupling and Injecting Domain Knowledge to Pre-trained Language Models' Memories. In Association for Computational Linguistics."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-023-01891-x"},{"key":"e_1_3_2_1_5_1","volume-title":"Doermann","author":"Gong Xuan","year":"2024","unstructured":"Xuan Gong, Shanglin Li, Yuxiang Bao, Barry Yao, Yawen Huang, Ziyan Wu, Baochang Zhang, Yefeng Zheng, and David S. Doermann. 2024. Federated Learning via Input-Output Collaborative Distillation. In AAAI Conference on Artificial Intelligence. AAAI Press, 22058--22066."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.121884"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-88682-2_21"},{"key":"e_1_3_2_1_8_1","volume-title":"Parameter-Efficient Transfer Learning with Diff Pruning","author":"Guo Demi","unstructured":"Demi Guo, Alexander M. Rush, and Yoon Kim. 2021. Parameter-Efficient Transfer Learning with Diff Pruning. In Association for Computational Linguistics."},{"key":"e_1_3_2_1_9_1","unstructured":"Hu Yao Zhu Chen He Xiaofei Cai Deng Guo Jia Chen Minghao. 2021. Reducing the teacher-student gap via spherical knowledge disitllation."},{"key":"e_1_3_2_1_10_1","volume-title":"International Conference on Learning Representations.","author":"He Junxian","year":"2022","unstructured":"Junxian He, Chunting Zhou, Xuezhe Ma, Taylor Berg-Kirkpatrick, and Graham Neubig. 2022. Towards a Unified View of Parameter-Efficient Transfer Learning. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_11_1","volume-title":"Deep Residual Learning for Image Recognition. In Conference on Computer Vision and Pattern Recognition. IEEE Computer Society, 770--778","author":"He Kaiming","year":"2016","unstructured":"Kaiming He, Xiangyu Zhang, Shaoqing Ren, and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In Conference on Computer Vision and Pattern Recognition. IEEE Computer Society, 770--778."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00201"},{"key":"e_1_3_2_1_13_1","volume-title":"Distilling the knowledge in a neural network. arXiv","author":"Hinton Geoffrey","year":"2015","unstructured":"Geoffrey Hinton, Oriol Vinyals, and Jeff Dean. 2015. Distilling the knowledge in a neural network. arXiv (2015)."},{"key":"e_1_3_2_1_14_1","volume-title":"Parameter-Efficient Transfer Learning for NLP. In International Conference on Machine Learning.","author":"Houlsby Neil","year":"2019","unstructured":"Neil Houlsby, Andrei Giurgiu, Stanislaw Jastrzebski, Bruna Morrone, Quentin de Laroussilhe, Andrea Gesmundo, Mona Attariyan, and Sylvain Gelly. 2019. Parameter-Efficient Transfer Learning for NLP. In International Conference on Machine Learning."},{"key":"e_1_3_2_1_15_1","unstructured":"Edward J. Hu Yelong Shen Phillip Wallis Zeyuan Allen-Zhu Yuanzhi Li Shean Wang Lu Wang and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In nternational Conference on Learning Representations."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/LGRS.2024.3402094"},{"key":"e_1_3_2_1_17_1","volume-title":"Conference on Neural Information Processing Systems","volume":"35","author":"Huang Tao","year":"2022","unstructured":"Tao Huang, Shan You, Fei Wang, Chen Qian, and Chang Xu. 2022. Knowledge distillation from a stronger teacher. Conference on Neural Information Processing Systems, Vol. 35 (2022), 33716--33727."},{"key":"e_1_3_2_1_18_1","first-page":"1","article-title":"Edge-Computing-Based Knowledge Distillation and Multitask Learning for Partial Discharge Recognition","volume":"73","author":"Ji Jinsheng","year":"2024","unstructured":"Jinsheng Ji, Zhou Shu, Hongqun Li, Kai Xian Lai, Minshan Lu, Guanlin Jiang, Wensong Wang, Yuanjin Zheng, and Xudong Jiang. 2024. Edge-Computing-Based Knowledge Distillation and Multitask Learning for Partial Discharge Recognition. IEEE Transactions on Instrumentation and Measurement, Vol. 73 (2024), 1--11.","journal-title":"IEEE Transactions on Instrumentation and Measurement"},{"key":"e_1_3_2_1_19_1","volume-title":"The Power of Scale for Parameter-Efficient Prompt Tuning. In Conference on Empirical Methods in Natural Language Processing.","author":"Lester Brian","year":"2021","unstructured":"Brian Lester, Rami Al-Rfou, and Noah Constant. 2021. The Power of Scale for Parameter-Efficient Prompt Tuning. In Conference on Empirical Methods in Natural Language Processing."},{"key":"e_1_3_2_1_20_1","volume-title":"Engineering and Management (Lecture Notes in Computer Science","volume":"453","author":"Li Haorong","year":"2023","unstructured":"Haorong Li, Zihao Chen, Jingtao Zhou, and Shuangyin Li. 2023. Reducing the Teacher-Student Gap via Elastic Student. In Knowledge Science, Engineering and Management (Lecture Notes in Computer Science, Vol. 14117). Springer, 442--453."},{"key":"e_1_3_2_1_21_1","volume-title":"Rethinking Feature-Based Knowledge Distillation for Face Recognition. In Conference on Computer Vision and Pattern Recognition. 20156--20165","author":"Li Jingzhi","year":"2023","unstructured":"Jingzhi Li, Zidong Guo, Hui Li, Seungju Han, Ji-won Baek, Min Yang, Ran Yang, and Sungjoo Suh. 2023. Rethinking Feature-Based Knowledge Distillation for Face Recognition. In Conference on Computer Vision and Pattern Recognition. 20156--20165."},{"key":"e_1_3_2_1_22_1","volume-title":"Prefix-Tuning: Optimizing Continuous Prompts for Generation","author":"Li Xiang Lisa","unstructured":"Xiang Lisa Li and Percy Liang. 2021. Prefix-Tuning: Optimizing Continuous Prompts for Generation. In Association for Computational Linguistics."},{"key":"e_1_3_2_1_23_1","volume-title":"Curriculum Temperature for Knowledge Distillation. In AAAI Conference on Artificial Intelligence. AAAI Press, 1504--1512","author":"Li Zheng","year":"2023","unstructured":"Zheng Li, Xiang Li, Lingfeng Yang, Borui Zhao, Renjie Song, Lei Luo, Jun Li, and Jian Yang. 2023. Curriculum Temperature for Knowledge Distillation. In AAAI Conference on Artificial Intelligence. AAAI Press, 1504--1512."},{"key":"e_1_3_2_1_24_1","volume-title":"International Conference on Machine Learning","volume":"202","author":"Liang Chen","year":"2023","unstructured":"Chen Liang, Simiao Zuo, Qingru Zhang, Pengcheng He, Weizhu Chen, and Tuo Zhao. 2023. Less is More: Task-aware Layer-wise Distillation for Language Model Compression. In International Conference on Machine Learning, Vol. 202. PMLR, 20852--20867."},{"key":"e_1_3_2_1_25_1","volume-title":"Function-Consistent Feature Distillation. In International Conference on Learning Representations. OpenReview.net.","author":"Liu Dongyang","year":"2023","unstructured":"Dongyang Liu, Meina Kan, Shiguang Shan, and Xilin Chen. 2023. Function-Consistent Feature Distillation. In International Conference on Learning Representations. OpenReview.net."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_53"},{"key":"e_1_3_2_1_27_1","unstructured":"Xiao Liu Yanan Zheng Zhengxiao Du Ming Ding Yujie Qian Zhilin Yang and Jie Tang. 2021. GPT Understands Too. CoRR (2021)."},{"key":"e_1_3_2_1_28_1","volume-title":"Sora: A Review on Background, Technology, Limitations, and Opportunities of Large Vision Models. CoRR","author":"Liu Yixin","year":"2024","unstructured":"Yixin Liu, Kai Zhang, Yuan Li, Zhiling Yan, Chujie Gao, Ruoxi Chen, Zhengqing Yuan, Yue Huang, Hanchi Sun, Jianfeng Gao, Lifang He, and Lichao Sun. 2024. Sora: A Review on Background, Technology, Limitations, and Opportunities of Large Vision Models. CoRR, Vol. abs\/2402.17177 (2024)."},{"key":"e_1_3_2_1_29_1","volume-title":"DeepFashion: Powering Robust Clothes Recognition and Retrieval with Rich Annotations. In Conference on Computer Vision and Pattern Recognition. IEEE Computer Society, 1096--1104","author":"Liu Ziwei","year":"2016","unstructured":"Ziwei Liu, Ping Luo, Shi Qiu, Xiaogang Wang, and Xiaoou Tang. 2016. DeepFashion: Powering Robust Clothes Recognition and Retrieval with Rich Annotations. In Conference on Computer Vision and Pattern Recognition. IEEE Computer Society, 1096--1104."},{"key":"e_1_3_2_1_30_1","volume-title":"Dual Relation Knowledge Distillation for Object Detection. In International Joint Conference on Artificial Intelligence. 1276--1284","author":"Ni Zhenliang","year":"2023","unstructured":"Zhenliang Ni, Fukui Yang, Shengzhao Wen, and Gang Zhang. 2023. Dual Relation Knowledge Distillation for Object Detection. In International Joint Conference on Artificial Intelligence. 1276--1284."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2023.3265382"},{"key":"e_1_3_2_1_33_1","volume-title":"Relational Knowledge Distillation. In Conference on Computer Vision and Pattern Recognition.","author":"Park Wonpyo","year":"2019","unstructured":"Wonpyo Park, Dongju Kim, Yan Lu, and Minsu Cho. 2019. Relational Knowledge Distillation. In Conference on Computer Vision and Pattern Recognition."},{"key":"e_1_3_2_1_34_1","volume-title":"Learning Deep Representations with Probabilistic Knowledge Transfer. In European Conference on Computer Vision (Lecture Notes in Computer Science","volume":"299","author":"Passalis Nikolaos","year":"2018","unstructured":"Nikolaos Passalis and Anastasios Tefas. 2018. Learning Deep Representations with Probabilistic Knowledge Transfer. In European Conference on Computer Vision (Lecture Notes in Computer Science, Vol. 11215). Springer, 283--299."},{"key":"e_1_3_2_1_35_1","volume-title":"AdapterFusion: Non-Destructive Task Composition for Transfer Learning. In Conference of the European Chapter of the Association for Computational Linguistics.","author":"Pfeiffer Jonas","year":"2021","unstructured":"Jonas Pfeiffer, Aishwarya Kamath, Andreas R\u00fcckl\u00e9, Kyunghyun Cho, and Iryna Gurevych. 2021. AdapterFusion: Non-Destructive Task Composition for Transfer Learning. In Conference of the European Chapter of the Association for Computational Linguistics."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2022.105560"},{"key":"e_1_3_2_1_37_1","volume-title":"International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"10357","author":"Touvron Hugo","year":"2021","unstructured":"Hugo Touvron, Matthieu Cord, Matthijs Douze, Francisco Massa, Alexandre Sablayrolles, and Herv\u00e9 J\u00e9gou. 2021. Training data-efficient image transformers & distillation through attention. In International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 139). PMLR, 10347--10357."},{"key":"e_1_3_2_1_38_1","volume-title":"Similarity-Preserving Knowledge Distillation. In International Conference on Computer Vision. IEEE, 1365--1374","author":"Tung Frederick","year":"2019","unstructured":"Frederick Tung and Greg Mori. 2019. Similarity-Preserving Knowledge Distillation. In International Conference on Computer Vision. IEEE, 1365--1374."},{"key":"e_1_3_2_1_39_1","volume-title":"Deep Hashing Network for Unsupervised Domain Adaptation. In Conference on Computer Vision and Pattern Recognition. IEEE Computer Society, 5385--5394","author":"Venkateswara Hemanth","year":"2017","unstructured":"Hemanth Venkateswara, Jose Eusebio, Shayok Chakraborty, and Sethuraman Panchanathan. 2017. Deep Hashing Network for Unsupervised Domain Adaptation. In Conference on Computer Vision and Pattern Recognition. IEEE Computer Society, 5385--5394."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2023.101583"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.121671"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i12.26699"},{"key":"e_1_3_2_1_43_1","volume-title":"Yuille","author":"Wang Yan","year":"2017","unstructured":"Yan Wang, Lingxi Xie, Ya Zhang, Wenjun Zhang, and Alan L. Yuille. 2017. Deep Collaborative Learning for Visual Recognition. CoRR, Vol. abs\/1703.01229 (2017)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109995"},{"key":"e_1_3_2_1_45_1","volume-title":"Joint Detection and Identification Feature Learning for Person Search. In Conference on Computer Vision and Pattern Recognition. IEEE Computer Society, 3376--3385","author":"Xiao Tong","year":"2017","unstructured":"Tong Xiao, Shuang Li, Bochao Wang, Liang Lin, and Xiaogang Wang. 2017. Joint Detection and Identification Feature Learning for Person Search. In Conference on Computer Vision and Pattern Recognition. IEEE Computer Society, 3376--3385."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-023-16551-y"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCDS.2022.3172937"},{"key":"e_1_3_2_1_48_1","first-page":"1","article-title":"Feature-Based Knowledge Distillation for Infrared Small Target Detection","volume":"21","author":"Xue Jinglei","year":"2024","unstructured":"Jinglei Xue, Jianan Li, Yuqi Han, Ze Wang, Chenwei Deng, and Tingfa Xu. 2024. Feature-Based Knowledge Distillation for Infrared Small Target Detection. Geoscience and Remote Sensing Letters, Vol. 21 (2024), 1--5.","journal-title":"Geoscience and Remote Sensing Letters"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3160328"},{"key":"e_1_3_2_1_50_1","volume-title":"BitFit: Simple Parameter-efficient Fine-tuning for Transformer-based Masked Language-models","author":"Zaken Elad Ben","unstructured":"Elad Ben Zaken, Yoav Goldberg, and Shauli Ravfogel. 2022. BitFit: Simple Parameter-efficient Fine-tuning for Transformer-based Masked Language-models. In Association for Computational Linguistics."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10758-023-09679-1"},{"key":"e_1_3_2_1_52_1","volume-title":"PAD: Self-Supervised Pre-Training with Patchwise-Scale Adapter for Infrared Images. CoRR","author":"Zhang Tao","year":"2023","unstructured":"Tao Zhang, Kun Ding, Jinyong Wen, Yu Xiong, Zeyu Zhang, Shiming Xiang, and Chunhong Pan. 2023. PAD: Self-Supervised Pre-Training with Patchwise-Scale Adapter for Infrared Images. CoRR, Vol. abs\/2312.08192 (2023)."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00454"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01165"},{"key":"e_1_3_2_1_55_1","volume-title":"Scalable Person Re-identification: A Benchmark. In International Conference on Computer Vision. IEEE Computer Society, 1116--1124","author":"Zheng Liang","year":"2015","unstructured":"Liang Zheng, Liyue Shen, Lu Tian, Shengjin Wang, Jingdong Wang, and Qi Tian. 2015. Scalable Person Re-identification: A Benchmark. In International Conference on Computer Vision. IEEE Computer Society, 1116--1124."},{"key":"e_1_3_2_1_56_1","volume-title":"Serial or Parallel? Plug-able Adapter for multilingual machine translation. CoRR","author":"Zhu Yaoming","year":"2021","unstructured":"Yaoming Zhu, Jiangtao Feng, Chengqi Zhao, Mingxuan Wang, and Lei Li. 2021. Serial or Parallel? Plug-able Adapter for multilingual machine translation. CoRR, Vol. abs\/2104.08154 (2021). gr"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681462","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681462","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:57:47Z","timestamp":1750294667000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681462"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":55,"alternative-id":["10.1145\/3664647.3681462","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681462","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}