{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T05:03:32Z","timestamp":1781586212696,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1145\/3731715.3733441","type":"proceedings-article","created":{"date-parts":[[2025,6,25]],"date-time":"2025-06-25T18:29:43Z","timestamp":1750876183000},"page":"303-311","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Taming Vision-Language Models for Federated Foundation Models on Heterogeneous Medical Imaging Modalities"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-5035-1626","authenticated-orcid":false,"given":"Lulu","family":"Feng","sequence":"first","affiliation":[{"name":"School of Computer Science and Technology, Xidian University, Xi'an, Shaanxi, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9992-2264","authenticated-orcid":false,"given":"Shengchao","family":"Chen","sequence":"additional","affiliation":[{"name":"Australian Artificial Intelligence Institute, University of Technology Sydney, Sydney, nsw, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,30]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Hisham Cholakkal, Mubarak Shah, Ming-Hsuan Yang, and Fahad Shahbaz Khan.","author":"Awais Muhammad","year":"2025","unstructured":"Muhammad Awais, Muzammal Naseer, Salman Khan, Rao Muhammad Anwer, Hisham Cholakkal, Mubarak Shah, Ming-Hsuan Yang, and Fahad Shahbaz Khan. 2025. Foundation Models Defining a New Era in Vision: a Survey and Outlook. IEEE Transactions on Pattern Analysis and Machine Intelligence (2025)."},{"key":"e_1_3_2_1_2_1","unstructured":"Guangyi Chen Weiran Yao Xiangchen Song Xinyue Li Yongming Rao and Kun Zhang. 2022. Prompt learning with optimal transport for vision-language models. (2022)."},{"key":"e_1_3_2_1_3_1","volume-title":"Foundation models for weather and climate data understanding: A comprehensive survey. arXiv preprint arXiv:2312.03014","author":"Chen Shengchao","year":"2023","unstructured":"Shengchao Chen, Guodong Long, Jing Jiang, Dikai Liu, and Chengqi Zhang. 2023a. Foundation models for weather and climate data understanding: A comprehensive survey. arXiv preprint arXiv:2312.03014 (2023)."},{"key":"e_1_3_2_1_4_1","first-page":"84897","article-title":"Personalized adapter for large meteorology model on devices: Towards weather foundation models","volume":"37","author":"Chen Shengchao","year":"2024","unstructured":"Shengchao Chen, Guodong Long, Jing Jiang, and Chengqi Zhang. 2024a. Personalized adapter for large meteorology model on devices: Towards weather foundation models. Advances in Neural Information Processing Systems, Vol. 37 (2024), 84897--84943.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i15.33739"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2023\/393"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2024\/638"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/JBHI.2023.3247949"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2024.111694"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1001\/jama.2018.5630"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3401777"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_13_1","volume-title":"Openprompt: An open-source framework for prompt-learning. arXiv preprint arXiv:2111.01998","author":"Ding Ning","year":"2021","unstructured":"Ning Ding, Shengding Hu, Weilin Zhao, Yulin Chen, Zhiyuan Liu, Hai-Tao Zheng, and Maosong Sun. 2021. Openprompt: An open-source framework for prompt-learning. arXiv preprint arXiv:2111.01998 (2021)."},{"key":"e_1_3_2_1_14_1","volume-title":"International colloquium on automata, languages, and programming","author":"Dwork Cynthia","unstructured":"Cynthia Dwork. 2006. Differential privacy. In International colloquium on automata, languages, and programming. Springer, 1--12."},{"key":"e_1_3_2_1_15_1","volume-title":"Regulation (EU)","volume":"679","author":"GDPR","year":"2016","unstructured":"GDPR GDPR. 2016. General data protection regulation. Regulation (EU), Vol. 679 (2016)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583518"},{"key":"e_1_3_2_1_17_1","volume-title":"Promptfl: Let federated participants cooperatively learn prompts instead of models-federated learning in age of foundation model","author":"Guo Tao","year":"2023","unstructured":"Tao Guo, Song Guo, Junxiao Wang, Xueyang Tang, and Wenchao Xu. 2023b. Promptfl: Let federated participants cooperatively learn prompts instead of models-federated learning in age of foundation model. IEEE Transactions on Mobile Computing (2023)."},{"key":"e_1_3_2_1_18_1","volume-title":"Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685","author":"Hu Edward J","year":"2021","unstructured":"Edward J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2021. Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685 (2021)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00746"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01832"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"e_1_3_2_1_22_1","first-page":"429","article-title":"Federated optimization in heterogeneous networks","volume":"2","author":"Li Tian","year":"2020","unstructured":"Tian Li, Anit Kumar Sahu, Manzil Zaheer, Maziar Sanjabi, Ameet Talwalkar, and Virginia Smith. 2020. Federated optimization in heterogeneous networks. Proceedings of Machine learning and systems, Vol. 2 (2020), 429--450.","journal-title":"Proceedings of Machine learning and systems"},{"key":"e_1_3_2_1_23_1","volume-title":"FedFMS: Exploring Federated Foundation Models for Medical Image Segmentation. In International Conference on Medical Image Computing and Computer-Assisted Intervention. Springer, 283--293","author":"Liu Yuxi","year":"2024","unstructured":"Yuxi Liu, Guibo Luo, and Yuesheng Zhu. 2024. FedFMS: Exploring Federated Foundation Models for Medical Image Segmentation. In International Conference on Medical Image Computing and Computer-Assisted Intervention. Springer, 283--293."},{"key":"e_1_3_2_1_24_1","volume-title":"Fedclip: Fast generalization and personalization for clip in federated learning. arXiv preprint arXiv:2302.13485","author":"Lu Wang","year":"2023","unstructured":"Wang Lu, Xixu Hu, Jindong Wang, and Xing Xie. 2023. Fedclip: Fast generalization and personalization for clip in federated learning. arXiv preprint arXiv:2302.13485 (2023)."},{"key":"e_1_3_2_1_25_1","unstructured":"Brendan McMahan Eider Moore Daniel Ramage Seth Hampson and Blaise Aguera y Arcas. 2017. Communication-efficient learning of deep networks from decentralized data. In Artificial intelligence and statistics. PMLR 1273--1282."},{"key":"e_1_3_2_1_26_1","first-page":"30590","article-title":"Federated Learning from Vision-Language Foundation Models: Theoretical Analysis and Method","volume":"37","author":"Pan Bikang","year":"2025","unstructured":"Bikang Pan, Wei Huang, and Ye Shi. 2025. Federated Learning from Vision-Language Foundation Models: Theoretical Analysis and Method. Advances in Neural Information Processing Systems, Vol. 37 (2025), 30590--30623.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_27_1","volume-title":"The Twelfth International Conference on Learning Representations.","author":"Qiu Chen","year":"2024","unstructured":"Chen Qiu, Xingyu Li, Chaithanya Kumar Mummadi, Madan Ravi Ganesh, Zhenzhen Li, Lu Peng, and Wan-Yi Lin. 2024. Federated text-driven prompt generation for vision-language models. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_28_1","volume-title":"International conference on machine learning. PMLR, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748--8763."},{"key":"e_1_3_2_1_29_1","volume-title":"Federated distillation for medical image classification: Towards trustworthy computer-aided diagnosis. arXiv preprint arXiv:2407.02261","author":"Ren Sufen","year":"2024","unstructured":"Sufen Ren, Yule Hu, Shengchao Chen, and Guanjun Wang. 2024. Federated distillation for medical image classification: Towards trustworthy computer-aided diagnosis. arXiv preprint arXiv:2407.02261 (2024)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i13.29416"},{"key":"e_1_3_2_1_31_1","volume-title":"Cross-domain federated adaptive prompt tuning for clip. arXiv preprint arXiv:2211.07864","author":"Su Shangchao","year":"2022","unstructured":"Shangchao Su, Mingzhao Yang, Bin Li, and Xiangyang Xue. 2022. Cross-domain federated adaptive prompt tuning for clip. arXiv preprint arXiv:2211.07864, Vol. 3 (2022)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00257"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11517-024-03101-3"},{"key":"e_1_3_2_1_34_1","volume-title":"Cooperative hardware-prompt learning for snapshot compressive imaging. arXiv preprint arXiv:2306.01176","author":"Wang Jiamian","year":"2023","unstructured":"Jiamian Wang, Zongliang Wu, Yulun Zhang, Xin Yuan, Tao Lin, and Zhiqiang Tao. 2023b. Cooperative hardware-prompt learning for snapshot compressive imaging. arXiv preprint arXiv:2306.01176 (2023)."},{"key":"e_1_3_2_1_35_1","volume-title":"Multi-Modal One-Shot Federated Ensemble Learning for Medical Data with Vision Large Language Model. arXiv preprint arXiv:2501.03292","author":"Wang Naibo","year":"2025","unstructured":"Naibo Wang, Yuchen Deng, Shichen Fan, Jianwei Yin, and See-Kiong Ng. 2025. Multi-Modal One-Shot Federated Ensemble Learning for Medical Data with Vision Large Language Model. arXiv preprint arXiv:2501.03292 (2025)."},{"key":"e_1_3_2_1_36_1","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision. 3032--3042","author":"Wang Zhengbo","year":"2023","unstructured":"Zhengbo Wang, Jian Liang, Ran He, Nan Xu, Zilei Wang, and Tieniu Tan. 2023a. Improving zero-shot generalization for clip with synthesized prompts. In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 3032--3042."},{"key":"e_1_3_2_1_37_1","volume-title":"Communication-efficient federated learning via knowledge distillation. Nature communications","author":"Wu Chuhan","year":"2022","unstructured":"Chuhan Wu, Fangzhao Wu, Lingjuan Lyu, Yongfeng Huang, and Xing Xie. 2022. Communication-efficient federated learning via knowledge distillation. Nature communications, Vol. 13, 1 (2022), 2032."},{"key":"e_1_3_2_1_38_1","volume-title":"FACMIC: Federated Adaptative CLIP Model for Medical Image Classification. In International Conference on Medical Image Computing and Computer-Assisted Intervention. Springer, 531--541","author":"Wu Yihang","year":"2024","unstructured":"Yihang Wu, Christian Desrosiers, and Ahmad Chaddad. 2024. FACMIC: Federated Adaptative CLIP Model for Medical Image Classification. In International Conference on Medical Image Computing and Computer-Assisted Intervention. Springer, 531--541."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01755"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41597-022-01721-8"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00653"},{"key":"e_1_3_2_1_42_1","volume-title":"On the challenges and perspectives of foundation models for medical image analysis. Medical image analysis","author":"Zhang Shaoting","year":"2024","unstructured":"Shaoting Zhang and Dimitris Metaxas. 2024. On the challenges and perspectives of foundation models for medical image analysis. Medical image analysis, Vol. 91 (2024), 102996."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","unstructured":"Ce Zhou Qian Li Chen Li Jun Yu Yixin Liu Guangjing Wang Kai Zhang Cheng Ji Qiben Yan Lifang He et al. 2024. A comprehensive survey on pretrained foundation models: A history from bert to chatgpt. International Journal of Machine Learning and Cybernetics (2024) 1--65.","DOI":"10.1007\/s13042-024-02443-6"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01653-1"},{"key":"e_1_3_2_1_46_1","volume-title":"When foundation model meets federated learning: Motivations, challenges, and future directions. arXiv preprint arXiv:2306.15546","author":"Zhuang Weiming","year":"2023","unstructured":"Weiming Zhuang, Chen Chen, and Lingjuan Lyu. 2023. When foundation model meets federated learning: Motivations, challenges, and future directions. arXiv preprint arXiv:2306.15546 (2023)."}],"event":{"name":"ICMR '25: International Conference on Multimedia Retrieval","location":"Chicago IL USA","acronym":"ICMR '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2025 International Conference on Multimedia Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3731715.3733441","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T04:09:05Z","timestamp":1755749345000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731715.3733441"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":46,"alternative-id":["10.1145\/3731715.3733441","10.1145\/3731715"],"URL":"https:\/\/doi.org\/10.1145\/3731715.3733441","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]},"assertion":[{"value":"2025-06-30","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}