{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T00:59:31Z","timestamp":1784941171891,"version":"3.55.0"},"reference-count":24,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,7,15]],"date-time":"2024-07-15T00:00:00Z","timestamp":1721001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,7,15]],"date-time":"2024-07-15T00:00:00Z","timestamp":1721001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100006190","name":"Research and Development","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006190","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004912","name":"Sichuan University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004912","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,7,15]]},"DOI":"10.1109\/icme57554.2024.10687740","type":"proceedings-article","created":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T17:24:16Z","timestamp":1727717056000},"page":"1-6","source":"Crossref","is-referenced-by-count":3,"title":["An Empirical Study of Parameter Efficient Fine-tuning on Vision-Language Pre-train Model"],"prefix":"10.1109","author":[{"given":"Yuxin","family":"Tian","sequence":"first","affiliation":[{"name":"Sichuan University,College of Computer Science,Chengdu,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mouxing","family":"Yang","sequence":"additional","affiliation":[{"name":"Sichuan University,College of Computer Science,Chengdu,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yunfan","family":"Li","sequence":"additional","affiliation":[{"name":"Sichuan University,College of Computer Science,Chengdu,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dayiheng","family":"Liu","sequence":"additional","affiliation":[{"name":"Sichuan University,College of Computer Science,Chengdu,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xingzhang","family":"Ren","sequence":"additional","affiliation":[{"name":"Peking University,School of Software and Microelectronics,Beijing,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xi","family":"Peng","sequence":"additional","affiliation":[{"name":"Sichuan University and Engineering Research Center of Machine Learning and Industry Intelligence, Ministry of Education,College of Computer Science,Chengdu,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiancheng","family":"Lv","sequence":"additional","affiliation":[{"name":"Sichuan University and Engineering Research Center of Machine Learning and Industry Intelligence, Ministry of Education,College of Computer Science,Chengdu,China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"23318","article-title":"OFA: Unifying Architectures, Tasks, and Modalities Through a Simple Sequenceto-Sequence Learning Framework","volume-title":"Proc. of ICML","author":"Wang"},{"key":"ref2","first-page":"8748","article-title":"Learning Transferable Visual Models From Natural Language Supervision","volume-title":"Proc. of ICML","author":"Radford"},{"key":"ref3","first-page":"9694","article-title":"Align before Fuse: Vision and Language Representation Learning with Momentum Distillation","volume-title":"Proc. of NeurIPS","author":"Li"},{"key":"ref4","first-page":"12888","article-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","volume-title":"Proc. of ICML","author":"Li"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"ref7","first-page":"2790","article-title":"Parameter-efficient transfer learning for NLP","volume-title":"Proc. of ICML","author":"Houlsby"},{"key":"ref8","article-title":"Towards a Unified View of Parameter-Efficient Transfer Learning","volume-title":"Proc. of ICLR","author":"He"},{"key":"ref9","article-title":"LoRA: Low-Rank Adaptation of Large Language Models","volume-title":"Proc. of ICLR","author":"Hu"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01653-1"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01832"},{"key":"ref12","article-title":"LLaMA-Adapter: Efficient Fine-tuning of Language Models with Zero-init Attention","volume":"abs\/2303.16199","author":"Zhang","year":"2023","journal-title":"CoRR"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00281"},{"key":"ref14","article-title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","volume":"abs\/2304.14178","author":"Ye","year":"2023","journal-title":"CoRR"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.168"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00516"},{"key":"ref17","article-title":"Benchmarking Robustness of Adaptation Methods on Pre-trained Vision-Language Models","volume":"abs\/2306.02080","author":"Chen","year":"2023","journal-title":"CoRR"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.568"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.488"},{"key":"ref20","first-page":"4171","article-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","volume-title":"Proc. of NAACL","author":"Devlin"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-018-1116-0"},{"key":"ref22","article-title":"Microsoft COCO Captions: Data Collection and Evaluation Server","volume":"abs\/1504.00325","author":"Chen","year":"2015","journal-title":"CoRR"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"ref24","first-page":"8681","article-title":"PALM: Pre-training an Autoencoding&Autoregressiive Language Model for Context-conditioned Generation","volume-title":"Proc. of EMNLP","author":"Bi"}],"event":{"name":"2024 IEEE International Conference on Multimedia and Expo (ICME)","location":"Niagara Falls, ON, Canada","start":{"date-parts":[[2024,7,15]]},"end":{"date-parts":[[2024,7,19]]}},"container-title":["2024 IEEE International Conference on Multimedia and Expo (ICME)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10685847\/10687354\/10687740.pdf?arnumber=10687740","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T06:37:10Z","timestamp":1727764630000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10687740\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,15]]},"references-count":24,"URL":"https:\/\/doi.org\/10.1109\/icme57554.2024.10687740","relation":{},"subject":[],"published":{"date-parts":[[2024,7,15]]}}}