{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T21:46:01Z","timestamp":1781819161985,"version":"3.54.5"},"publisher-location":"Singapore","reference-count":39,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819530540","type":"print"},{"value":"9789819530557","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T00:00:00Z","timestamp":1763078400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T00:00:00Z","timestamp":1763078400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-3055-7_5","type":"book-chapter","created":{"date-parts":[[2025,11,13]],"date-time":"2025-11-13T04:07:23Z","timestamp":1763006843000},"page":"53-64","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["MMtuning: An Advanced Multi-adapter Framework for\u00a0Efficient Multimodal Large Language Models Fine-Tuning"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-0721-2763","authenticated-orcid":false,"given":"Li","family":"Qiao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4323-7166","authenticated-orcid":false,"given":"Haowen","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8318-239X","authenticated-orcid":false,"given":"Kazunori","family":"Sugiura","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-3090-1240","authenticated-orcid":false,"given":"Keren","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5601-7261","authenticated-orcid":false,"given":"Jinglu","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,11,14]]},"reference":[{"key":"5_CR1","doi-asserted-by":"crossref","unstructured":"Aghajanyan, A., Gupta, S., Zettlemoyer, L.: Intrinsic dimensionality explains the effectiveness of language model fine-tuning. In: Zong, C., Xia, F., Li, W., Navigli, R. (eds.) Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), pp. 7319\u20137328. Association for Computational Linguistics (2021)","DOI":"10.18653\/v1\/2021.acl-long.568"},{"key":"5_CR2","unstructured":"Alayrac, J.B., et al.: Flamingo: a visual language model for few-shot learning. In: Proceedings of the 36th Conference on Neural Information Processing Systems (2022)"},{"key":"5_CR3","unstructured":"Alexey, D., et al.: An image is worth 16x16 words: transformers for image recognition at scale. In: Proceedings of The 7th International Conference on Learning Representations (2021)"},{"key":"5_CR4","unstructured":"Bai, J., et al.: Qwen-vl: A versatile vision-language model for understanding, localization, text reading, and beyond. arXiv preprint arXiv:2308.12966 [cs.CV] (2023)"},{"key":"5_CR5","unstructured":"Caccia, L., Ponti, E., Su, Z., Pereira, M., Le\u00a0Roux, N., Sordoni, A.: Multi-head adapter routing for cross-task generalization. In: Proceedings of the 37th International Conference on Neural Information Processing Systems. NIPS \u201923, Curran Associates Inc. (2023)"},{"key":"5_CR6","unstructured":"Du, N., et al.: GLaM: Efficient scaling of language models with mixture-of-experts. In: Proceedings of the 39th International Conference on Machine Learning. vol.\u00a0162, pp. 5547\u20135569 (2022)"},{"key":"5_CR7","unstructured":"Eigen, D., Ranzato, M., Sutskever, I.: Learning factored representations in a deep mixture of experts. arXiv preprint arXiv:1312.4314v3 [cs.LG] (2014)"},{"key":"5_CR8","first-page":"1","volume":"23","author":"W Fedus","year":"2022","unstructured":"Fedus, W., Shazeer, N., Clark, A.: Switch transformers: scaling to trillion parameter models with simple and efficient sparsity. J. Mach. Learn. Res. 23, 1\u201339 (2022)","journal-title":"J. Mach. Learn. Res."},{"key":"5_CR9","unstructured":"Geoffrey, H., Oriol, V., Jeffrey, D.: Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531v1 [stat.ML] (2015)"},{"key":"5_CR10","unstructured":"Houlsby, N., et al.: Parameter-efficient transfer learning for NLP. In: Proceedings of the 36th International Conference on Machine Learning. vol.\u00a097, pp. 2790\u20132799 (2019)"},{"key":"5_CR11","unstructured":"Hu, E., et al.: Lora: low-rank adaptation of large lan-guage models. In: Proceedings of the International Conference on Learning Representations (2022)"},{"key":"5_CR12","unstructured":"Hyeon-Woo, N., Ye-Bin, M., Oh, T.H.: Fedpara: low-rank hadamard product for communication-efficient federated learning. In: Proceedings of the International Conference on Learning Representations (2022)"},{"key":"5_CR13","unstructured":"Lepikhin, D., Chen, D., Shazeer, N., Chen, Z.: Gshard: scaling giant models with condi-tional computation and automatic sharding. In: Proceedings of the International Conference on Learning Representations (2021)"},{"key":"5_CR14","doi-asserted-by":"crossref","unstructured":"Lester, B., Al-Rfou, R., Constant, N.: The power of scale for parameter-efficient prompt tuning. In: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 3045\u20133059 (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"5_CR15","unstructured":"Li, B., Zhang, Y., Chen, L., Wang, J., Yang, J., Liu, Z.: Otter: a multi-modal model with in-context instruction tuning (2023)"},{"key":"5_CR16","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: Proceedings of the 40th International Conference on Machine Learning, pp. 19730\u201319742 (2023)"},{"key":"5_CR17","doi-asserted-by":"crossref","unstructured":"Li, Y., Wang, H., Duan, Y., Zhang, J., Li, X.: A closer look at the explainability of contrastive language-image pre-training (2024)","DOI":"10.1016\/j.patcog.2025.111409"},{"key":"5_CR18","unstructured":"Liu, H., et al.: Few-shot parameter-efficient fine-tuning is better and cheaper than in-context learning. In: Oh, A.H., Agarwal, A., Belgrave, D., Cho, K. (eds.) Advances in Neural Information Processing Systems (2022)"},{"key":"5_CR19","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. In: Thirty-seventh Conference on Neural Information Processing Systems. vol.\u00a036, pp. 34892\u201334916 (2023)"},{"key":"5_CR20","unstructured":"Liu, T., Blondel, M., Ruiz, C.R., Puigcerver, J.: Routers in vision mixture of experts: an empirical study. Trans. Mach. Learn. Res. (2024)"},{"key":"5_CR21","doi-asserted-by":"crossref","unstructured":"Liu, X., et al.: P-tuning: prompt tuning can be comparable to fine-tuning across scales and tasks. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics. vol.\u00a002, pp. 61\u201368 (2022)","DOI":"10.18653\/v1\/2022.acl-short.8"},{"key":"5_CR22","unstructured":"Liu, X., et al.: Gpt understands, too. arXiv preprint arXiv:2103.10385v2 [cs.CL] (2023)"},{"key":"5_CR23","unstructured":"Lou, Y., Xue, F., Zheng, Z., You, Y.: Cross-token modeling with conditional computation. arXiv preprint arXiv:2109.02008v3 [cs.LG] (2022)"},{"key":"5_CR24","unstructured":"Lu, P., et al.: Learn to explain: Multimodal reasoning via thought chains for science question answering. In: Proceedings of the 36th Conference on Neural Information Processing Systems (2022)"},{"key":"5_CR25","unstructured":"Ponti, E., Sordoni, A., Reddy, S.: Combining modular skills in multitask learning. arXiv preprint arXiv:2202.13914v2 [cs.LG] (2022)"},{"key":"5_CR26","doi-asserted-by":"crossref","unstructured":"Qin, Y., et al.: Exploring universal intrinsic task subspace for few-shot learning via prompt tuning. IEEE\/ACM Trans. Audio Speech Lang. Proc. 32, 3631\u20133643 (2024)","DOI":"10.1109\/TASLP.2024.3430545"},{"issue":"140","key":"5_CR27","first-page":"1","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel, C., et al.: Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res. 21(140), 1\u201367 (2020)","journal-title":"J. Mach. Learn. Res."},{"key":"5_CR28","unstructured":"Shazeer, N.M., et al.: Outrageously large neural networks: the sparsely-gated mixture-of-experts layer. In: Proceedings of the International Conference on Learning Representations (2017)"},{"key":"5_CR29","unstructured":"Wang, H., Sun, T., Ji, K., Wang, J., Fan, C., Gu, J.: Orchmoe: efficient multi-adapter learning with task-skill synergy (2024)"},{"key":"5_CR30","unstructured":"Wang, H., et al.: Customizable combination of parameter-efficient modules for multi-task learning. In: Proceedings of the Twelfth International Conference on Learning Representations (2024)"},{"key":"5_CR31","unstructured":"Wang, Y., Lin, Y., Zeng, X., Zhang, G.: Multilora: democratizing lora for better multi-task learning. arXiv preprint arXiv:2311.11501v1 [cs.LG] (2023)"},{"key":"5_CR32","doi-asserted-by":"crossref","unstructured":"Xiang, L., Li, Liang, P.: Prefix-tuning: optimizing continuous prompts for generation. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (ACL\/IJCNLP), pp. 4582\u20134597 (2021)","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"5_CR33","unstructured":"Zadouri, T., \u00dcst \u00dcn, A., Ahmadian, A., Ermis, B., Locatelli, A., Hooker, S.: Pushing mixture of experts to the limit: extremely parameter efficient moe for instruction tuning. In: Proceedings of the International Conference on Learning Representations (2024)"},{"key":"5_CR34","doi-asserted-by":"crossref","unstructured":"Zhang, D.Z., et al.: Mm-LLMS: recent advances in multimodal large language models. arXiv preprint arXiv:2401.13601v4 [cs.CL] (2024)","DOI":"10.18653\/v1\/2024.findings-acl.738"},{"key":"5_CR35","unstructured":"Zhang, Q., et al.: Adalora: adaptive budget allocation for parameter-efficient fine-tuning. In: Proceedings of the 11th International Conference on Learning Representations (ICLR) (2023)"},{"key":"5_CR36","doi-asserted-by":"crossref","unstructured":"Zhang, S., et al.: Instruction tuning for large language models: a survey (2024)","DOI":"10.18653\/v1\/2024.findings-acl.341"},{"key":"5_CR37","unstructured":"Zhu, D., Chen, J., Shen, X., Li, X., Elhoseiny, M.: MiniGPT-4: enhancing vision-language understanding with advanced large language models. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"5_CR38","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Groth, O., Bernstein, M., Fei-Fei, L.: Visual7w: grounded question answering in images. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 4995\u20135004 (2016)","DOI":"10.1109\/CVPR.2016.540"},{"key":"5_CR39","unstructured":"Zuo, S., et al.: Taming sparsely activated transformer with stochastic experts. In: Proceedings of the International Conference on Learning Representations (2022)"}],"container-title":["Lecture Notes in Computer Science","Knowledge Science, Engineering and Management"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-3055-7_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,13]],"date-time":"2025-11-13T04:07:31Z","timestamp":1763006851000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-3055-7_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,14]]},"ISBN":["9789819530540","9789819530557"],"references-count":39,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-3055-7_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,14]]},"assertion":[{"value":"14 November 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"KSEM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Knowledge Science, Engineering and Management","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Macao","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 August 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7 August 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ksem2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ksem2025.scimeeting.cn\/en\/web\/index\/27434","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}