{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T15:45:48Z","timestamp":1784821548846,"version":"3.55.0"},"reference-count":243,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s11263-026-02864-6","type":"journal-article","created":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T02:48:06Z","timestamp":1780454886000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Parameter-Efficient Fine-Tuning for Pre-Trained Vision Models: A Survey and Benchmark"],"prefix":"10.1007","volume":"134","author":[{"given":"Yi","family":"Xin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianjiang","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Siqi","family":"Luo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuntao","family":"Du","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qi","family":"Qin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haoxing","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kangrui","family":"Cen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yangfan","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bin","family":"Fu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuewen","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junjun","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaokang","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guangtao","family":"Zhai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ming-Hsuan","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6377-4730","authenticated-orcid":false,"given":"Xiaohong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,3]]},"reference":[{"key":"2864_CR1","unstructured":"adaptation, O.-l.T.-f. (2024). Online-lora: Task-free online continual learning via low rank adaptation. IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)."},{"key":"2864_CR2","doi-asserted-by":"crossref","unstructured":"Agiza, A., Neseem, M., & Reda, S. (2024). Mtlora: Low-rank adaptation approach for efficient multi-task learning. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52733.2024.01533"},{"issue":"2020","key":"2864_CR3","doi-asserted-by":"publisher","DOI":"10.1016\/j.dib.2019.104863","volume":"28","author":"W Al-Dhabyani","year":"2020","unstructured":"Al-Dhabyani, W., Gomaa, M., Khaled, H., & Fahmy, A. (2020). Dataset of breast ultrasound images. Data in brief, 28(2020), Article 104863.","journal-title":"Data in brief"},{"issue":"1","key":"2864_CR4","doi-asserted-by":"publisher","first-page":"4128","DOI":"10.1038\/s41467-022-30695-9","volume":"13","author":"M Antonelli","year":"2022","unstructured":"Antonelli, M., Reinke, A., Bakas, S., Farahani, K., Kopp-Schneider, A., Landman, B. A., Litjens, G., Menze, B., Ronneberger, O., Summers, R. M., & Van Ginneken, B. (2022). The medical segmentation decathlon. Nature communications, 13(1), 4128.","journal-title":"Nature communications"},{"key":"2864_CR5","doi-asserted-by":"crossref","unstructured":"Baek, S., Lee, S., Jo, H., Choi, H., & Min, D. (2025). Tadformer: Task-adaptive dynamic transformer for efficient multi-task learning. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52734.2025.01384"},{"key":"2864_CR6","unstructured":"Bahng, H., Jahanian, A., Sankaranarayanan, S., & Isola, P. (2022). Exploring visual prompts for adapting large-scale models. arXiv:2203.17274."},{"key":"2864_CR7","doi-asserted-by":"crossref","unstructured":"Bandara, W.G.C., & Patel, V.M. (2024). Attention prompt tuning: Parameter-efficient adaptation of pre-trained models for action recognition. International Conference on Automatic Face and Gesture Recognition (FG).","DOI":"10.1109\/FG59268.2024.10581865"},{"key":"2864_CR8","doi-asserted-by":"crossref","unstructured":"Basu, S., Hu, S., Massiceti, D., & Feizi, S. (2024). Strong baselines for parameter efficient few-shot fine-tuning. Proceedings of the AAAI Conference on Artificial Intelligence (AAAI).","DOI":"10.1609\/aaai.v38i10.28978"},{"key":"2864_CR9","doi-asserted-by":"crossref","unstructured":"Basu, S., Hu, S., Massiceti, D., & Feizi, S. (2024). Strong baselines for parameter-efficient few-shot fine-tuning. Proceedings of the AAAI Conference on Artificial Intelligence (AAAI),.","DOI":"10.1609\/aaai.v38i10.28978"},{"key":"2864_CR10","unstructured":"Beattie, C., Leibo, J.Z., Teplyashin, D., Ward, T., Wainwright, M., K\u00fcttler, H., Lefrancq, A., Green, S., Vald\u00e9s, V., Sadik, A., & Schrittwieser, J. (2016) Deepmind lab. arXiv:1612.03801"},{"issue":"1","key":"2864_CR11","doi-asserted-by":"publisher","first-page":"131","DOI":"10.1109\/TPAMI.2024.3461779","volume":"47","author":"H Bi","year":"2024","unstructured":"Bi, H., Feng, Y., Diao, W., Wang, P., Mao, Y., Fu, K., Wang, H., & Sun, X. (2024). Prompt-and-transfer: Dynamic class-aware enhancement for few-shot segmentation. IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 47(1), 131\u2013148.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI)"},{"key":"2864_CR12","unstructured":"Bu, Z., Wang, Y.-X., Zha, S., & Karypis, G. (2022). Differentially private bias-term only fine-tuning of foundation models. Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"2864_CR13","doi-asserted-by":"crossref","unstructured":"Caron, M., Touvron, H., Misra, I., J\u00e9gou, H., Mairal, J., Bojanowski, P., & Joulin, A. (2021). Emerging properties in self-supervised vision transformers. Proceedings of the IEEE International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"2864_CR14","unstructured":"Carreira, J., Noland, E., Hillier, C., & Zisserman, A. (2019). A short note on the kinetics-700 human action dataset. arXiv:1907.06987"},{"key":"2864_CR15","doi-asserted-by":"publisher","first-page":"4951","DOI":"10.1016\/j.procs.2024.09.452","volume":"246","author":"S Chai","year":"2024","unstructured":"Chai, S., Jain, R. K., Teng, S., Liu, J., Li, Y., Tateyama, T., & Chen, Y.-W. (2024). Ladder fine-tuning approach for sam integrating complementary network. Procedia Computer Science, 246, 4951\u20134958.","journal-title":"Procedia Computer Science"},{"key":"2864_CR16","unstructured":"Chen, D. (2023). Aggregate, decompose, and fine-tune: A simple yet effective factor-tuning method for vision transformer. arXiv:2311.06749"},{"key":"2864_CR17","unstructured":"Chen, Z., Duan, Y., Wang, W., He, J., Lu, T., Dai, J., & Qiao, Y. (2023). Vision transformer adapter for dense predictions. Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"2864_CR18","unstructured":"Chen, T., Kornblith, S., Norouzi, M., & Hinton, G. (2020). A simple framework for contrastive learning of visual representations. Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"2864_CR19","doi-asserted-by":"crossref","unstructured":"Chen, D.-Y., Tennent, H., & Hsu, C.-W. (2024). Artadapter: Text-to-image style transfer using multi-level style encoder and explicit adaptation. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52733.2024.00823"},{"key":"2864_CR20","doi-asserted-by":"crossref","unstructured":"Chen, A., Yao, Y., Chen, P.-Y., Zhang, Y., & Liu, S. (2023). Understanding and improving visual prompting: A label-mapping perspective. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52729.2023.01834"},{"key":"2864_CR21","doi-asserted-by":"crossref","unstructured":"Chen, J., Yu, J., Ge, C., Yao, L., Xie, E., Wu, Y., Wang, Z., Kwok, J., Luo, P., Lu, H., & Li, Z. (2023). Pixart-$$alpha$$: Fast training of diffusion transformer for photorealistic text-to-image synthesis. Proceedings of the International Conference on Learning Representations (ICLR).","DOI":"10.1007\/978-3-031-73411-3_5"},{"key":"2864_CR22","doi-asserted-by":"crossref","unstructured":"Cheng, G., Han, J., & Lu, X. (2017). Remote sensing image scene classification: Benchmark and state of the art. Proceedings of the IEEE,.","DOI":"10.1109\/JPROC.2017.2675998"},{"key":"2864_CR23","doi-asserted-by":"publisher","first-page":"16664","DOI":"10.52202\/068431-1212","volume":"35","author":"S Chen","year":"2022","unstructured":"Chen, S., Ge, C., Tong, Z., Wang, J., Song, Y., Wang, J., & Luo, P. (2022). Adaptformer: Adapting vision transformers for scalable visual recognition. Advances in Neural Information Processing Systems (NeurIPS), 35, 16664\u201316678.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2864_CR24","doi-asserted-by":"crossref","unstructured":"Cimpoi, M., Maji, S., Kokkinos, I., Mohamed, S., & Vedaldi, A. (2014). Describing textures in the wild. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR.2014.461"},{"issue":"9","key":"2864_CR25","doi-asserted-by":"publisher","first-page":"10850","DOI":"10.1109\/TPAMI.2023.3261988","volume":"45","author":"F-A Croitoru","year":"2023","unstructured":"Croitoru, F.-A., Hondru, V., Ionescu, R. T., & Shah, M. (2023). Diffusion models in vision: A survey. IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 45(9), 10850\u201310869.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI)"},{"issue":"1","key":"2864_CR26","doi-asserted-by":"publisher","first-page":"33","DOI":"10.1007\/s11263-021-01531-2","volume":"130","author":"D Damen","year":"2022","unstructured":"Damen, D., Doughty, H., Farinella, G. M., Furnari, A., Kazakos, E., Ma, J., Moltisanti, D., Munro, J., Perrett, T., Price, W., & Wray, M. (2022). Rescaling egocentric vision: Collection, pipeline and challenges for epic-kitchens-100. International Journal of Computer Vision (IJCV), 130(1), 33\u201355.","journal-title":"International Journal of Computer Vision (IJCV)"},{"key":"2864_CR27","doi-asserted-by":"crossref","unstructured":"Das, R., Dukler, Y., Ravichandran, A., & Swaminathan, A. (2023). Learning expressive prompting with residuals for vision transformers. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52729.2023.00328"},{"issue":"10","key":"2864_CR28","doi-asserted-by":"publisher","first-page":"7854","DOI":"10.3390\/su15107854","volume":"15","author":"H Dastour","year":"2023","unstructured":"Dastour, H., & Hassan, Q. K. (2023). A comparison of deep transfer learning methods for land use and land cover classification. Sustainability, 15(10), 7854.","journal-title":"Sustainability"},{"key":"2864_CR29","unstructured":"Dataset, E. (2011). Novel datasets for fine-grained image categorization. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops (CVPR Workshops),."},{"key":"2864_CR30","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.-J., Li, K., & Fei-Fei, L. (2009). Imagenet: A large-scale hierarchical image database. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"2864_CR31","unstructured":"Deng, X., Fan, Q., Jin, X., Yang, L., & Wang, P. (2023). Selective feature adapter for dense vision transformers. arXiv:2310.01843."},{"key":"2864_CR32","doi-asserted-by":"crossref","unstructured":"Diao, H., Wan, B., Jia, X., Zhuge, Y., Zhang, Y., Lu, H., & Chen, L. (2024). Sherl: Synthesizing high accuracy and efficient memory for resource-limited transfer learning. In: Proceedings of the European Conference on Computer Vision (ECCV).","DOI":"10.1007\/978-3-031-72784-9_5"},{"key":"2864_CR33","doi-asserted-by":"crossref","unstructured":"Ding, N., Qin, Y., Yang, G., Wei, F., Yang, Z., Su, Y., Hu, S., Chen, Y., Chan, C.-M., Chen, W., & Yi, J. (2022). Delta tuning: A comprehensive study of parameter efficient methods for pre-trained language models. Nature Machine Intelligence,.","DOI":"10.21203\/rs.3.rs-1553541\/v1"},{"key":"2864_CR34","unstructured":"Dong, B., Zhou, P., Yan, S., & Zuo, W. (2023). Lpt: Long-tailed prompt tuning for image classification. Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"2864_CR35","doi-asserted-by":"publisher","first-page":"102056","DOI":"10.52202\/079017-3239","volume":"37","author":"W Dong","year":"2024","unstructured":"Dong, W., Sun, Y., Yang, Y., Zhang, X., Lin, Z., Yan, Q., Zhang, H., Wang, P., Yang, Y., & Shen, H. (2024). Efficient adaptation of pre-trained vision transformer via householder transformation. Advances in Neural Information Processing Systems (NeurIPS), 37, 102056\u2013102077.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2864_CR36","first-page":"52548","volume":"36","author":"W Dong","year":"2024","unstructured":"Dong, W., Yan, D., Lin, Z., & Wang, P. (2024). Efficient adaptation of large vision transformer via adapter re-composing. Advances in Neural Information Processing Systems (NeurIPS), 36, 52548\u201352567.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2864_CR37","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., & Uszkoreit, J. (2021). An image is worth 16x16 words: Transformers for image recognition at scale. Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"2864_CR38","unstructured":"Du, S., Zhang, G., Wang, K., Wang, Y., Yue, H., Zhang, G., Ding, E., Wang, J., Xu, Z., & Yuan, C. (2024). Alore: Efficient visual adaptation via aggregating low rank experts. arXiv:2412.08341"},{"key":"2864_CR39","doi-asserted-by":"crossref","unstructured":"Dutt, R., Bohdal, O., Sanchez, P., Tsaftaris, S., & Hospedales, T. (2025). Memcontrol: Mitigating memorization in diffusion models via automated parameter selection. In: IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV).","DOI":"10.1109\/WACV61041.2025.00441"},{"key":"2864_CR40","volume-title":"Krona: Parameter efficient tuning with kronecker adapter","author":"A Edalati","year":"2025","unstructured":"Edalati, A., Tahaei, M., Kobyzev, I., Nia, V. P., Clark, J. J., & Rezagholizadeh, M. (2025). Krona: Parameter efficient tuning with kronecker adapter. Enhancing LLM Performance: Efficacy, Fine-Tuning, and Inference Techniques."},{"issue":"1","key":"2864_CR41","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1007\/s11263-014-0733-5","volume":"111","author":"M Everingham","year":"2015","unstructured":"Everingham, M., Eslami, S. A., Van Gool, L., Williams, C. K., Winn, J., & Zisserman, A. (2015). The pascal visual object classes challenge: A retrospective. International Journal of Computer Vision (IJCV), 111(1), 98\u2013136.","journal-title":"International Journal of Computer Vision (IJCV)"},{"key":"2864_CR42","doi-asserted-by":"crossref","unstructured":"Fang, Y., Wang, W., Xie, B., Sun, Q., Wu, L., Wang, X., Huang, T., Wang, X., & Cao, Y. (2022). Eva: Exploring the limits of masked visual representation learning at scale. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52729.2023.01855"},{"key":"2864_CR43","doi-asserted-by":"crossref","unstructured":"Fang, Z., Wang, Y., Yi, R., & Ma, L. (2025). Dropout mixture low-rank adaptation for visual parameters-efficient fine-tuning. Proceedings of the European Conference on Computer Vision (ECCV),.","DOI":"10.1007\/978-3-031-72667-5_21"},{"key":"2864_CR44","first-page":"1","volume":"61","author":"L Fang","year":"2023","unstructured":"Fang, L., Kuang, Y., Liu, Q., Yang, Y., & Yue, J. (2023). Rethinking remote sensing pretrained model: Instance-aware visual prompting for remote sensing scene classification. IEEE Transactions on Geoscience and Remote Sensing (TGRS), 61, 1\u201313.","journal-title":"IEEE Transactions on Geoscience and Remote Sensing (TGRS)"},{"issue":"4","key":"2864_CR45","doi-asserted-by":"publisher","first-page":"594","DOI":"10.1109\/TPAMI.2006.79","volume":"28","author":"L Fei-Fei","year":"2006","unstructured":"Fei-Fei, L., Fergus, R., & Perona, P. (2006). One-shot learning of object categories. IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 28(4), 594\u2013611.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI)"},{"key":"2864_CR46","doi-asserted-by":"crossref","unstructured":"Fu, C.-L., Chen, Z.-C., Lee, Y.-R., & Lee, H.-y. (2022). Adapterbias: Parameter-efficient token-dependent representation shift for adapters in nlp tasks. Proceedings of the Annual Conference of the North American Chapter of the Association for Computational Linguistics (NAACL).","DOI":"10.18653\/v1\/2022.findings-naacl.199"},{"key":"2864_CR47","doi-asserted-by":"crossref","unstructured":"Fu, M., Zhu, K., & Wu, J. (2024). Dtl: Disentangled transfer learning for visual recognition. Proceedings of the AAAI Conference on Artificial Intelligence (AAAI),.","DOI":"10.1609\/aaai.v38i11.29096"},{"key":"2864_CR48","doi-asserted-by":"crossref","unstructured":"Fu, M., Zhu, K., & Wu, J. (2024). Dtl: Disentangled transfer learning for visual recognition. Proceedings of the AAAI Conference on Artificial Intelligence (AAAI).","DOI":"10.1609\/aaai.v38i11.29096"},{"key":"2864_CR49","doi-asserted-by":"crossref","unstructured":"Gandikota, R., Materzy\u0144ska, J., Zhou, T., Torralba, A., & Bau, D. (2024). Concept sliders: Lora adaptors for precise control in diffusion models. Proceedings of the European Conference on Computer Vision (ECCV),.","DOI":"10.1007\/978-3-031-73661-2_10"},{"key":"2864_CR50","unstructured":"Gangwar, N., Rangi, A., Deshmukh, R., Rahmanian, H., Dattatreya, Y., & Kani, N. (2025). Parameter-efficient multi-task learning via progressive task-specific adaptation. arXiv:2509.19602"},{"key":"2864_CR51","unstructured":"Gao, Y., Shi, X., Zhu, Y., Wang, H., Tang, Z., Zhou, X., Li, M., & Metaxas, D.N. (2022). Visual prompt tuning for test-time domain adaptation. arXiv:2210.04831"},{"key":"2864_CR52","doi-asserted-by":"crossref","unstructured":"Gao, Q., Zhao, C., Sun, Y., Xi, T., Zhang, G., Ghanem, B., & Zhang, J. (2023). A unified continual learning framework with general parameter-efficient tuning. Proceedings of the IEEE International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV51070.2023.01055"},{"key":"2864_CR53","doi-asserted-by":"crossref","unstructured":"Gebru, T., Krause, J., Wang, Y., Chen, D., Deng, J., & Fei-Fei, L. (2017). Fine-grained car detection for visual census estimation. Proceedings of the AAAI Conference on Artificial Intelligence (AAAI),.","DOI":"10.1609\/aaai.v31i1.11174"},{"issue":"11","key":"2864_CR54","doi-asserted-by":"publisher","first-page":"1231","DOI":"10.1177\/0278364913491297","volume":"32","author":"A Geiger","year":"2013","unstructured":"Geiger, A., Lenz, P., Stiller, C., & Urtasun, R. (2013). Vision meets robotics: The kitti dataset. The International Journal of Robotics Research, 32(11), 1231\u20131237.","journal-title":"The International Journal of Robotics Research"},{"key":"2864_CR55","doi-asserted-by":"crossref","unstructured":"Goyal, R., Ebrahimi\u00a0Kahou, S., Michalski, V., Materzynska, J., Westphal, S., Kim, H., Haenel, V., Fruend, I., Yianilos, P., Mueller-Freitag, M., & Hoppe, F. (2017). The something something video database for learning and evaluating visual common sense. Proceedings of the IEEE International Conference on Computer Vision (ICCV),.","DOI":"10.1109\/ICCV.2017.622"},{"key":"2864_CR56","unstructured":"Graham, B. (2015). Kaggle diabetic retinopathy detection competition report. University of Warwick."},{"key":"2864_CR57","doi-asserted-by":"crossref","unstructured":"Guo, X., Zheng, M., Hou, L., Gao, Y., Deng, Y., Wan, P., Zhang, D., Liu, Y., Hu, W., Zha, Z., & Huang, H. (2024). I2v-adapter: A general image-to-video adapter for diffusion models. ACM SIGGRAPH 2024 Conference Papers.","DOI":"10.1145\/3641519.3657407"},{"key":"2864_CR58","doi-asserted-by":"crossref","unstructured":"Han, R., & Tang, J. (2025). Straightforward layer-wise pruning for more efficient visual adaptation. Proceedings of the European Conference on Computer Vision (ECCV),.","DOI":"10.1007\/978-3-031-73220-1_14"},{"key":"2864_CR59","unstructured":"Han, Z., Gao, C., Liu, J., Zhang, J., & Zhang, S. Q. (2024). Parameter-efficient fine-tuning for large models: A comprehensive survey. arXiv:2403.14608"},{"key":"2864_CR60","doi-asserted-by":"crossref","unstructured":"Han, C., Wang, Q., Cui, Y., Cao, Z., Wang, W., Qi, S., & Liu, D. (2023). E2vpt: An effective and efficient approach for visual prompt tuning. Proceedings of the IEEE International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV51070.2023.01604"},{"key":"2864_CR61","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., & Girshick, R. (2022). Masked autoencoders are scalable vision learners. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"2864_CR62","doi-asserted-by":"crossref","unstructured":"He, K., Fan, H., Wu, Y., Xie, S., & Girshick, R. (2020). Momentum contrast for unsupervised visual representation learning. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"2864_CR63","doi-asserted-by":"crossref","unstructured":"He, X., Li, C., Zhang, P., Yang, J., & Wang, X.E. (2023). Parameter-efficient model adaptation for vision transformers. Proceedings of the AAAI Conference on Artificial Intelligence (AAAI),.","DOI":"10.1609\/aaai.v37i1.25160"},{"key":"2864_CR64","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR.2016.90"},{"issue":"7","key":"2864_CR65","doi-asserted-by":"publisher","first-page":"2217","DOI":"10.1109\/JSTARS.2019.2918242","volume":"12","author":"P Helber","year":"2019","unstructured":"Helber, P., Bischke, B., Dengel, A., & Borth, D. (2019). Eurosat: A novel dataset and deep learning benchmark for land use and land cover classification. IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing, 12(7), 2217\u20132226.","journal-title":"IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing"},{"key":"2864_CR66","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Basart, S., Mu, N., Kadavath, S., Wang, F., Dorundo, E., Desai, R., Zhu, T., Parajuli, S., Guo, M., & Song, D. (2021). The many faces of robustness: A critical analysis of out-of-distribution generalization. Proceedings of the IEEE International Conference on Computer Vision (ICCV),.","DOI":"10.1109\/ICCV48922.2021.00823"},{"key":"2864_CR67","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Zhao, K., Basart, S., Steinhardt, J., & Song, D. (2021). Natural adversarial examples. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR46437.2021.01501"},{"key":"2864_CR68","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., & Abbeel, P. (2020). Denoising diffusion probabilistic models. Advances in Neural Information Processing Systems (NeurIPS), 33, 6840\u20136851.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2864_CR69","doi-asserted-by":"crossref","unstructured":"Hong, L., Yan, S., Zhang, R., Li, W., Zhou, X., Guo, P., Jiang, K., Chen, Y., Li, J., Chen, Z., & Zhang, W. (2024). Onetracker: Unifying visual object tracking with foundation models and efficient tuning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52733.2024.01805"},{"key":"2864_CR70","unstructured":"Houlsby, N., Giurgiu, A., Jastrzebski, S., Morrone, B., De\u00a0Laroussilhe, Q., Gesmundo, A., Attariyan, M., & Gelly, S. (2019). Parameter-efficient transfer learning for nlp. Proceedings of the International Conference on Machine Learning (ICML)."},{"issue":"12","key":"2864_CR71","doi-asserted-by":"publisher","first-page":"8274","DOI":"10.1109\/TPAMI.2024.3401450","volume":"46","author":"Q Hou","year":"2024","unstructured":"Hou, Q., Lu, C.-Z., Cheng, M.-M., & Feng, J. (2024). Conv2former: A simple transformer-style convnet for visual recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, 46(12), 8274\u20138283.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2864_CR72","unstructured":"Hu, S., Liao, Z., & Xia, Y. (2022). Prosfda: Prompt learning based source-free domain adaptation for medical image segmentation. arXiv:2211.11514"},{"key":"2864_CR73","unstructured":"Hu, E.J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., & Chen, W. (2021). Lora: Low-rank adaptation of large language models. Proceedings of the International Conference on Learning Representations (ICLR),."},{"key":"2864_CR74","doi-asserted-by":"crossref","unstructured":"Huang, Q., Dong, X., Chen, D., Zhang, W., Wang, F., Hua, G., & Yu, N. (2023). Diversity-aware meta visual prompting. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52729.2023.01047"},{"key":"2864_CR75","doi-asserted-by":"crossref","unstructured":"Huang, L., Mao, J., Yi, J., Tao, Z., & Wang, Y. (2025). Cvpt: Cross visual prompt tuning. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV51701.2025.00087"},{"key":"2864_CR76","unstructured":"Huang, L., Wang, W., Wu, Z.-F., Shi, Y., Dou, H., Liang, C., Feng, Y., Liu, Y., & Zhou, J. (2024). In-context lora for diffusion transformers. arXiv:2410.23775"},{"key":"2864_CR77","doi-asserted-by":"crossref","unstructured":"Huang, J., Xu, Z., Liu, T., Liu, Y., Han, H., Yuan, K., & Li, X. (2025). Densely connected parameter-efficient tuning for referring image segmentation. In: Proceedings of the AAAI Conference on Artificial Intelligence (AAAI).","DOI":"10.1609\/aaai.v39i4.32380"},{"key":"2864_CR78","first-page":"1","volume":"62","author":"L Hu","year":"2024","unstructured":"Hu, L., Lu, W., Yu, H., Yin, D., Sun, X., & Fu, K. (2024). Tea: A training-efficient adapting framework for tuning foundation models in remote sensing. IEEE Transactions on Geoscience and Remote Sensing (TGRS), 62, 1\u201318.","journal-title":"IEEE Transactions on Geoscience and Remote Sensing (TGRS)"},{"key":"2864_CR79","first-page":"1","volume":"62","author":"L Hu","year":"2024","unstructured":"Hu, L., Yu, H., Lu, W., Yin, D., Sun, X., & Fu, K. (2024). Airs: Adapter in remote sensing for parameter-efficient transfer learning. IEEE Transactions on Geoscience and Remote Sensing (TGRS), 62, 1\u201318.","journal-title":"IEEE Transactions on Geoscience and Remote Sensing (TGRS)"},{"key":"2864_CR80","doi-asserted-by":"crossref","unstructured":"Jia, M., Tang, L., Chen, B.-C., Cardie, C., Belongie, S., Hariharan, B., & Lim, S.-N. (2022). Visual prompt tuning. Proceedings of the European Conference on Computer Vision (ECCV).","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"2864_CR81","unstructured":"Jia, C., Yang, Y., Xia, Y., Chen, Y.-T., Parekh, Z., Pham, H., Le, Q., Sung, Y.-H., Li, Z., & Duerig, T. (2021). Scaling up visual and vision-language representation learning with noisy text supervision. Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"2864_CR82","doi-asserted-by":"crossref","unstructured":"Jiang, Z., Chen, T., Chen, X., Cheng, Y., Zhou, L., Yuan, L., Awadallah, A., & Wang, Z. (2022) Dna: Improving few-shot transfer learning with low-rank decomposition and alignment. Proceedings of the European Conference on Computer Vision (ECCV),.","DOI":"10.1007\/978-3-031-20044-1_14"},{"key":"2864_CR83","unstructured":"Jiang, Z., Mao, C., Huang, Z., Lv, Y., Zhao, D., & Zhou, J. (2023). Rethinking efficient tuning methods from a unified perspective. arXiv:2303.00690"},{"key":"2864_CR84","doi-asserted-by":"crossref","unstructured":"Jiang, F., Wang, S., & Gong, X. (2024). Task-conditional adapter for multi-task dense prediction. Proceedings of the ACM Conference on Multimedia (MM),.","DOI":"10.1145\/3664647.3681581"},{"key":"2864_CR85","doi-asserted-by":"crossref","unstructured":"Jie, S., & Deng, Z.-H. (2023). Fact: Factor-tuning for lightweight adaptation on vision transformer. Proceedings of the AAAI Conference on Artificial Intelligence (AAAI),.","DOI":"10.1609\/aaai.v37i1.25187"},{"key":"2864_CR86","doi-asserted-by":"crossref","unstructured":"Jie, S., Deng, Z.-H., Chen, S., & Jin, Z. (2024). Convolutional bypasses are better vision transformer adapters. European Conference on Artificial Intelligence (ECAI).","DOI":"10.3233\/FAIA240489"},{"key":"2864_CR87","doi-asserted-by":"crossref","unstructured":"Jie, S., Tang, Y., Guo, J., Deng, Z.-H., Han, K., & Wang, Y. (2024). Token compensator: Altering inference cost of vision transformer without re-tuning. In: Proceedings of the European Conference on Computer Vision (ECCV),.","DOI":"10.1007\/978-3-031-72640-8_5"},{"key":"2864_CR88","doi-asserted-by":"crossref","unstructured":"Johnson, J., Hariharan, B., Van Der\u00a0Maaten, L., Fei-Fei, L., Lawrence\u00a0Zitnick, C., & Girshick, R. (2017). Clevr: A diagnostic dataset for compositional language and elementary visual reasoning. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR.2017.215"},{"key":"2864_CR89","unstructured":"Kay, W., Carreira, J., Simonyan, K., Zhang, B., Hillier, C., Vijayanarasimhan, S., Viola, F., Green, T., Back, T., Natsev, P., & Suleyman, M. (2017). The kinetics human action video dataset. arXiv:1705.06950"},{"issue":"10s","key":"2864_CR90","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3505244","volume":"54","author":"S Khan","year":"2022","unstructured":"Khan, S., Naseer, M., Hayat, M., Zamir, S. W., Khan, F. S., & Shah, M. (2022). Transformers in vision: A survey. ACM Computing Surveys (CSUR), 54(10s), 1\u201341.","journal-title":"ACM Computing Surveys (CSUR)"},{"key":"2864_CR91","doi-asserted-by":"crossref","unstructured":"Khoba, P.K., Wang, Z., Arora, C., Baktashmotlagh, M.: Peftdiff: Diffusion-guided transferability estimation for parameter-efficient fine-tuning. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV) (2025)","DOI":"10.1109\/ICCV51701.2025.00143"},{"key":"2864_CR92","doi-asserted-by":"crossref","unstructured":"Kim, K., Park, J., Kim, J., Kwon, H., & Sohn, K. (2025). Faster parameter-efficient tuning with token redundancy reduction. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52734.2025.02810"},{"key":"2864_CR93","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2024.106414","volume":"178","author":"S Kim","year":"2023","unstructured":"Kim, S., Yang, H., Kim, Y., Hong, Y., & Park, E. (2023). Hydra: Multi-head low-rank adaptation for parameter efficient fine-tuning. Neural Networks, 178, Article 106414.","journal-title":"Neural Networks"},{"key":"2864_CR94","unstructured":"Kingma, D.P., & Welling, M. (2013) Auto-encoding variational bayes. arXiv:1312.6114"},{"key":"2864_CR95","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.-Y., & Doll\u00e1r, P. (2023). Segment anything. Proceedings of the IEEE International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"2864_CR96","first-page":"1","volume":"62","author":"Y Kong","year":"2024","unstructured":"Kong, Y., Cheng, Y., Chen, Y., & Wang, X. (2024). Joint classification of hyperspectral image and lidar data based on spectral prompt tuning. IEEE Transactions on Geoscience and Remote Sensing (TGRS), 62, 1\u201312.","journal-title":"IEEE Transactions on Geoscience and Remote Sensing (TGRS)"},{"key":"2864_CR97","doi-asserted-by":"crossref","unstructured":"Kornblith, S., Shlens, J., & Le, Q.V. (2019). Do better imagenet models transfer better? Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR.2019.00277"},{"key":"2864_CR98","unstructured":"Krizhevsky, A., & Hinton, G. (2009). Learning multiple layers of features from tiny images. https:\/\/www.cs.toronto.edu\/kriz\/learning-features-2009-TR.pdf"},{"key":"2864_CR99","doi-asserted-by":"crossref","unstructured":"Kuehne, H., Jhuang, H., Garrote, E., Poggio, T., & Serre, T. (2011). Hmdb: a large video database for human motion recognition. Proceedings of the IEEE International Conference on Computer Vision (ICCV),.","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"2864_CR100","doi-asserted-by":"crossref","unstructured":"LeCun, Y., Huang, F.J., & Bottou, L. (2004). Learning methods for generic object recognition with invariance to pose and lighting. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR.2004.1315150"},{"key":"2864_CR101","doi-asserted-by":"publisher","first-page":"8152","DOI":"10.52202\/075280-0357","volume":"36","author":"T Lei","year":"2023","unstructured":"Lei, T., Bai, J., Brahma, S., Ainslie, J., Lee, K., Zhou, Y., Du, N., Zhao, V., Wu, Y., Li, B., & Zhang, Y. (2023). Conditional adapters: Parameter-efficient transfer learning with fast inference. Advances in Neural Information Processing Systems (NeurIPS), 36, 8152\u20138172.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2864_CR102","doi-asserted-by":"crossref","unstructured":"Li, X.L., & Liang, P. (2021). Prefix-tuning: Optimizing continuous prompts for generation. Proceedings of the Annual Meeting of the Association for Computational Linguistics (ACL).","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"2864_CR103","doi-asserted-by":"crossref","unstructured":"Li, Y., Li, Y., & Vasconcelos, N. (2018). Resound: Towards action recognition without representation bias. Proceedings of the European Conference on Computer Vision (ECCV),.","DOI":"10.1007\/978-3-030-01231-1_32"},{"key":"2864_CR104","unstructured":"Li, X., Tramer, F., Liang, P., & Hashimoto, T. (2012). Large language models can be strong differentially private learners. Proceedings of the International Conference on Learning Representations (ICLR),."},{"key":"2864_CR105","doi-asserted-by":"crossref","unstructured":"Liang, Y.-S., & Li, W.-J. (2024). Inflora: Interference-free low-rank adaptation for continual learning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52733.2024.02231"},{"key":"2864_CR106","unstructured":"Liang, J., Huang, W., Guo, X., Wan, G., Du, B., & Ye, M. (2025). Thanora: Task heterogeneity-aware multi-task low-rank adaptation. arXiv:2505.18640"},{"key":"2864_CR107","doi-asserted-by":"publisher","first-page":"109","DOI":"10.52202\/068431-0009","volume":"35","author":"D Lian","year":"2022","unstructured":"Lian, D., Zhou, D., Feng, J., & Wang, X. (2022). Scaling & shifting your features: A new baseline for efficient model tuning. Advances in Neural Information Processing Systems (NeurIPS), 35, 109\u2013123.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2864_CR108","doi-asserted-by":"crossref","unstructured":"Lin, L., Fan, H., Zhang, Z., Wang, Y., Xu, Y., & Ling, H. (2024). Tracking meets lora: Faster training, larger model, stronger performance. Proceedings of the European Conference on Computer Vision (ECCV),.","DOI":"10.1007\/978-3-031-73232-4_17"},{"key":"2864_CR109","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., & Zitnick, C.L. (2014). Microsoft coco: Common objects in context. Proceedings of the European Conference on Computer Vision (ECCV),.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2864_CR110","unstructured":"Lin, W., Wu, Z., Yang, W., Huang, M., Huang, J., & Jin, L. (2023). Hierarchical side-tuning for vision transformers. arXiv:2310.05393 (2023)"},{"key":"2864_CR111","doi-asserted-by":"crossref","unstructured":"Lin, J., Yin, H., Ping, W., Molchanov, P., Shoeybi, M., & Han, S. (2024). Vila: On pre-training for visual language models. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52733.2024.02520"},{"key":"2864_CR112","doi-asserted-by":"crossref","unstructured":"Liu, J., Chang, Y., & Wu, Y. (2025). R-lora: Random initialization of multi-head lora for multi-task learning. Arxiv.","DOI":"10.18653\/v1\/2025.findings-emnlp.35"},{"key":"2864_CR113","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., & Guo, B. (2021). Swin transformer: Hierarchical vision transformer using shifted windows. Proceedings of the IEEE International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"2864_CR114","doi-asserted-by":"crossref","unstructured":"Liu, X., Liu, T., Huang, S., Xin, Y., Hu, Y., Qin, L., Wang, D., Wu, Y., & Chen, H. (2025). M2ist: Multi-modal interactive side-tuning for efficient referring expression comprehension. IEEE Transactions on Circuits and Systems for Video Technology (TCSVT).","DOI":"10.1109\/TCSVT.2025.3551766"},{"key":"2864_CR115","unstructured":"Liu, T., Liu, X., Shi, L., Xu, Z., Hu, Y., Huang, S., Xin, Y., Zhong, B., & Wang, D. (2024). Sparse-tuning: Adapting vision transformers with efficient fine-tuning and inference. arXiv:2405.14700"},{"key":"2864_CR116","doi-asserted-by":"crossref","unstructured":"Liu, Z., Ning, J., Cao, Y., Wei, Y., Zhang, Z., Lin, S., & Hu, H. (2022). Video swin transformer. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"2864_CR117","doi-asserted-by":"crossref","unstructured":"Liu, W., Shen, X., Pun, C.-M., & Cun, X. (2023). Explicit visual prompting for low-level structure segmentations. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52729.2023.01862"},{"key":"2864_CR118","doi-asserted-by":"crossref","unstructured":"Liu, D., Xin, Y., Zhao, S., Zhuo, L., Lin, W., Li, X., Qin, Q., Zhai, G., Liu, X., Li, H., & Gao, P. (2026). Lumina-mgpt: Illuminate flexible photorealistic text-to-image generation with multimodal generative pretraining. International Journal of Computer Vision (IJCV),.","DOI":"10.1007\/s11263-026-02795-2"},{"key":"2864_CR119","doi-asserted-by":"crossref","unstructured":"Liu, Z., Xu, K., Su, B., Zou, X., Peng, Y., & Zhou, J. (2025). Stop: Integrated spatial-temporal dynamic prompting for video understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52734.2025.01286"},{"key":"2864_CR120","doi-asserted-by":"publisher","first-page":"36889","DOI":"10.52202\/068431-2673","volume":"35","author":"Y-C Liu","year":"2022","unstructured":"Liu, Y.-C., Ma, C.-Y., Tian, J., He, Z., & Kira, Z. (2022). Polyhistor: Parameter-efficient multi-task adaptation for dense vision tasks. Advances in Neural Information Processing Systems (NeurIPS), 35, 36889\u201336901.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"issue":"2","key":"2864_CR121","doi-asserted-by":"publisher","first-page":"1489","DOI":"10.1109\/TPAMI.2022.3164083","volume":"45","author":"Y Li","year":"2022","unstructured":"Li, Y., Yao, T., Pan, Y., & Mei, T. (2022). Contextual transformer networks for visual recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 45(2), 1489\u20131500.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI)"},{"issue":"3","key":"2864_CR122","doi-asserted-by":"publisher","first-page":"190","DOI":"10.1109\/MGRS.2025.3541952","volume":"13","author":"S Lu","year":"2025","unstructured":"Lu, S., Guo, J., Zimmer-Dauphinee, J. R., Nieusma, J. M., Wang, X., VanValkenburgh, P., Wernke, S. A., & Huo, Y. (2025). Vision foundation models in remote sensing: A survey. IEEE Geoscience and Remote Sensing Magazine, 13(3), 190\u2013215.","journal-title":"IEEE Geoscience and Remote Sensing Magazine"},{"key":"2864_CR123","unstructured":"Luo, G., Huang, M., Zhou, Y., Sun, X., Jiang, G., Wang, Z., & Ji, R. (2023). Towards efficient visual adaption via structural re-parameterization. arXiv:2302.08106"},{"key":"2864_CR124","doi-asserted-by":"crossref","unstructured":"Luo, L., Li, S., Ren, D., Wang, Q., Zhu, P., & Hu, Q. (2025). Decoupled multi-predictor optimization for inference-efficient model tuning. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV),.","DOI":"10.1109\/ICCV51701.2025.00346"},{"key":"2864_CR125","unstructured":"Luo, J., Pang, Z., Zhang, Y., Wang, T., Wang, L., Dang, B., Lao, J., Wang, J., Chen, J., Tan, Y., & Li, Y. (2024). Skysensegpt: A fine-grained instruction tuning dataset and model for remote sensing vision-language understanding. arXiv:2406.10100"},{"key":"2864_CR126","doi-asserted-by":"crossref","unstructured":"Luo, S., Yang, H., Xin, Y., Yi, M., Wu, G., Zhai, G., & Liu, X. (2025). Tr-pts: Task-relevant parameter and token selection for efficient tuning. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV),.","DOI":"10.1109\/ICCV51701.2025.00415"},{"key":"2864_CR127","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2024.103447","volume":"101","author":"X Luo","year":"2025","unstructured":"Luo, X., Fu, J., Zhong, Y., Liu, S., Han, B., Astaraki, M., Bendazzoli, S., Toma-Dasu, I., Ye, Y., Chen, Z., & Xia, Y. (2025). Segrap 2023: A benchmark of organs-at-risk and gross tumor volume segmentation for radiotherapy planning of nasopharyngeal carcinoma. Medical Image Analysis, 101, Article 103447.","journal-title":"Medical Image Analysis"},{"key":"2864_CR128","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2024.106854","volume":"181","author":"H Luo","year":"2025","unstructured":"Luo, H., Hu, W., Wei, Y., He, J., & Yu, M. (2025). Hirmtl: Hierarchical multi-task learning for dense scene understanding. Neural Networks, 181, Article 106854.","journal-title":"Neural Networks"},{"key":"2864_CR129","unstructured":"Ly, S.T., & Nguyen, H.V. (2024). Enhancing parameter-efficient fine-tuning of vision transformers through frequency-based adaptation. Arxiv."},{"key":"2864_CR130","unstructured":"Mahabadi, R.K., Ruder, S., Dehghani, M., & Henderson, J. (2021). Parameter-efficient multi-task fine-tuning for transformers via shared hypernetworks. Proceedings of the Annual Meeting of the Association for Computational Linguistics (ACL),."},{"key":"2864_CR131","doi-asserted-by":"crossref","unstructured":"Mai, Z., Zhang, P., Tu, C.-H., Chen, H.-Y., Zhang, L., & Chao, W.-L. (2025). Lessons learned from a unifying empirical study of parameter-efficient transfer learning (petl) in visual recognition. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52734.2025.01383"},{"key":"2864_CR132","doi-asserted-by":"crossref","unstructured":"Mantri, K.S.I., Sch\u00f6nlieb, C.-B., Ribeiro, B., Baskin, C., & Eliasof, M. (2025). Ditask: Multi-task fine-tuning with diffeomorphic transformations. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52734.2025.02348"},{"key":"2864_CR133","doi-asserted-by":"crossref","unstructured":"Marouf, I.E., Tartaglione, E., & Lathuili\u00e8re, S. (2024). Mini but mighty: Finetuning vits with mini adapters. Proceedings of the IEEE Winter Conference on Applications of Computer Vision (WACV).","DOI":"10.1109\/WACV57701.2024.00175"},{"key":"2864_CR134","unstructured":"Matthey, L., Higgins, I., Hassabis, D., & Lerchner, A. (2017) dSprites: Disentanglement testing Sprites dataset. https:\/\/github.com\/deepmind\/dsprites-dataset\/"},{"issue":"10","key":"2864_CR135","doi-asserted-by":"publisher","first-page":"6695","DOI":"10.1109\/TPAMI.2021.3100536","volume":"44","author":"J Ma","year":"2021","unstructured":"Ma, J., Zhang, Y., Gu, S., Zhu, C., Ge, C., Zhang, Y., An, X., Wang, C., Wang, Q., Liu, X., & Cao, S. (2021). Abdomenct-1k: Is abdominal organ segmentation a solved problem? IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 44(10), 6695\u20136714.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI)"},{"key":"2864_CR136","doi-asserted-by":"crossref","unstructured":"Mercea, O.-B., Gritsenko, A., Schmid, C., & Arnab, A. (2024). Time-memory-and parameter-efficient visual adaptation. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52733.2024.00529"},{"key":"2864_CR137","doi-asserted-by":"crossref","unstructured":"Miao, Z., Chen, W., & Qiu, Q. (2025). Coeff-tuning: A graph filter subspace view for tuning attention-based large models. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52734.2025.01876"},{"key":"2864_CR138","doi-asserted-by":"crossref","unstructured":"Mou, C., Wang, X., Xie, L., Wu, Y., Zhang, J., Qi, Z., & Shan, Y. (2024). T2i-adapter: Learning adapters to dig out more controllable ability for text-to-image diffusion models. Proceedings of the AAAI Conference on Artificial Intelligence (AAAI).","DOI":"10.1609\/aaai.v38i5.28226"},{"key":"2864_CR139","unstructured":"Netzer, Y., Wang, T., Coates, A., Bissacco, A., Wu, B., & Ng, A.Y. (2011). Reading digits in natural images with unsupervised feature learning. Advances in Neural Information Processing Systems Workshops (NeurIPS Workshops),."},{"issue":"6","key":"2864_CR140","doi-asserted-by":"publisher","first-page":"4653","DOI":"10.1109\/TCSVT.2023.3327605","volume":"34","author":"X Nie","year":"2023","unstructured":"Nie, X., Ni, B., Chang, J., Meng, G., Huo, C., Xiang, S., & Tian, Q. (2023). Pro-tuning: Unified prompt tuning for vision tasks. IEEE Transactions on Circuits and Systems for Video Technology (TCSVT), 34(6), 4653\u20134667.","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology (TCSVT)"},{"key":"2864_CR141","doi-asserted-by":"crossref","unstructured":"Nilsback, M.-E., & Zisserman, A. (2008). Automated flower classification over a large number of classes. 2008 Sixth Indian conference on computer vision, graphics & image processing,.","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"2864_CR142","unstructured":"Nowak, A.I., Mercea, O.-B., Arnab, A., Pfeiffer, J., Dauphin, Y., & Evci, U. (2024). Towards optimal adapter placement for efficient transfer learning. arXiv:2410.15858"},{"key":"2864_CR143","doi-asserted-by":"publisher","first-page":"26462","DOI":"10.52202\/068431-1919","volume":"35","author":"J Pan","year":"2022","unstructured":"Pan, J., Lin, Z., Zhu, X., Shao, J., & Li, H. (2022). St-adapter: Parameter-efficient image-to-video transfer learning. Advances in Neural Information Processing Systems(NeurIPS), 35, 26462\u201326477.","journal-title":"Advances in Neural Information Processing Systems(NeurIPS)"},{"key":"2864_CR144","doi-asserted-by":"crossref","unstructured":"Park, S., & Byun, H. (2024). Fair-vpt: Fair visual prompt tuning for image classification. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52733.2024.01166"},{"key":"2864_CR145","doi-asserted-by":"crossref","unstructured":"Parkhi, O.M., Vedaldi, A., Zisserman, A., & Jawahar, C. (2012). Cats and dogs. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR.2012.6248092"},{"key":"2864_CR146","doi-asserted-by":"crossref","unstructured":"Pei, W., Xia, T., Chen, F., Li, J., Tian, J., & Lu, G. (2024). Sa$$^2$$vp: Spatially aligned-and-adapted visual prompt. Proceedings of the AAAI Conference on Artificial Intelligence (AAAI).","DOI":"10.1609\/aaai.v38i5.28243"},{"key":"2864_CR147","doi-asserted-by":"crossref","unstructured":"Peng, Z., Xu, Z., Zeng, Z., Xie, L., Tian, Q., & Shen, W. (2024). Parameter efficient fine-tuning via cross block orchestration for segment anything model. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52733.2024.00359"},{"key":"2864_CR148","unstructured":"Podell, D., English, Z., Lacey, K., Blattmann, A., Dockhorn, T., M\u00fcller, J., Penna, J., & Rombach, R. (2023). Sdxl: Improving latent diffusion models for high-resolution image synthesis. Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"2864_CR149","first-page":"1","volume":"63","author":"X Pu","year":"2025","unstructured":"Pu, X., & Xu, F. (2025). Low-rank adaption on transformer-based oriented object detector for satellite onboard processing of remote sensing images. IEEE Transactions on Geoscience and Remote Sensing (TGRS), 63, 1\u201313.","journal-title":"IEEE Transactions on Geoscience and Remote Sensing (TGRS)"},{"key":"2864_CR150","doi-asserted-by":"crossref","unstructured":"Qin, Q., Zhuo, L., Xin, Y., Du, R., Li, Z., Fu, B., Lu, Y., Li, X., Liu, D., Zhu, X., & Beddow, W. (2025). Lumina-image 2.0: A unified and efficient image generative framework. Proceedings of the IEEE International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV51701.2025.01863"},{"key":"2864_CR151","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., & Krueger, G. (2021). Learning transferable visual models from natural language supervision. Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"2864_CR152","unstructured":"Recht, B., Roelofs, R., Schmidt, L., & Shankar, V. (2019). Do imagenet classifiers generalize to imagenet? Proceedings of the International Conference on Machine Learning (ICML),."},{"key":"2864_CR153","unstructured":"Ridnik, T., Ben-Baruch, E., Noy, A., & Zelnik-Manor, L. (2021). Imagenet-21k pretraining for the masses. Advances in Neural Information Processing Systems (NeurIPS),."},{"key":"2864_CR154","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., & Ommer, B. (2022). High-resolution image synthesis with latent diffusion models. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"2864_CR155","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., & Brox, T. (2015). U-net: Convolutional networks for biomedical image segmentation. Medical Image Computing and Computer Assisted Intervention (MICCAI).","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"2864_CR156","doi-asserted-by":"crossref","unstructured":"Ruan, J., Gao, J., Xie, M., Xiang, S., Yu, Z., Liu, T., Fu, Y., & Qu, X. (2024). Gist: Improving parameter efficient fine-tuning via knowledge interaction. Proceedings of the ACM Conference on Multimedia (MM),.","DOI":"10.1145\/3664647.3680843"},{"issue":"1","key":"2864_CR157","doi-asserted-by":"publisher","DOI":"10.1016\/j.jksuci.2024.101921","volume":"36","author":"M Sahu","year":"2024","unstructured":"Sahu, M., Dash, R., Mishra, S. K., Humayun, M., Alfayad, M., & Assiri, M. (2024). A deep transfer learning model for green environment security analysis in smart city. Journal of King Saud University-Computer and Information Sciences, 36(1), Article 101921.","journal-title":"Journal of King Saud University-Computer and Information Sciences"},{"key":"2864_CR158","doi-asserted-by":"crossref","unstructured":"Shang, C., Li, M., Zhang, Y., Chen, Z., Wu, J., Gu, F., Lu, Y., & Cheung, Y.-m. (2025). Pro-vpt: Distribution-adaptive visual prompt tuning via prompt relocation. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV51701.2025.00153"},{"issue":"6","key":"2864_CR159","doi-asserted-by":"publisher","first-page":"3613","DOI":"10.1007\/s11263-024-02274-6","volume":"133","author":"R Shao","year":"2025","unstructured":"Shao, R., Wu, T., Nie, L., & Liu, Z. (2025). Deepfake-adapter: Dual-level adapter for deepfake detection. International Journal of Computer Vision (IJCV), 133(6), 3613\u20133628.","journal-title":"International Journal of Computer Vision (IJCV)"},{"key":"2864_CR160","unstructured":"Sharma, M., Fantacci, C., Zhou, Y., Koppula, S., Heess, N., Scholz, J., & Aytar, Y. (2023). Lossless adaptation of pretrained vision models for robotic manipulation. Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"2864_CR161","doi-asserted-by":"crossref","unstructured":"Sohn, K., Chang, H., Lezama, J., Polania, L., Zhang, H., Hao, Y., Essa, I., & Jiang, L. (2023). Visual prompt tuning for generative transfer learning. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52729.2023.01900"},{"key":"2864_CR162","unstructured":"Soomro, K., Zamir, A.R., & Shah, M. (2012). Ucf101: A dataset of 101 human actions classes from videos in the wild. arXiv:1212.0402"},{"key":"2864_CR163","doi-asserted-by":"crossref","unstructured":"Strudel, R., Garcia, R., Laptev, I., & Schmid, C. (2021). Segmenter: Transformer for semantic segmentation. Proceedings of the IEEE International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV48922.2021.00717"},{"key":"2864_CR164","doi-asserted-by":"crossref","unstructured":"Sun, G., Mendieta, M., Luo, J., Wu, S., & Chen, C. (2023). Fedperfix: Towards partial model personalization of vision transformers in federated learning. Proceedings of the IEEE International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV51070.2023.00460"},{"key":"2864_CR165","doi-asserted-by":"publisher","first-page":"12991","DOI":"10.52202\/068431-0944","volume":"35","author":"Y-L Sung","year":"2022","unstructured":"Sung, Y.-L., Cho, J., & Bansal, M. (2022). Lst: Ladder side-tuning for parameter and memory efficient transfer learning. Advances in Neural Information Processing Systems (NeurIPS), 35, 12991\u201313005.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2864_CR166","doi-asserted-by":"crossref","unstructured":"Tamirisa, R., Xie, C., Bao, W., Zhou, A., Arel, R., & Shamsian, A. (2024). Fedselect: Personalized federated learning with customized selection of parameters for fine-tuning. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52733.2024.02264"},{"key":"2864_CR167","doi-asserted-by":"crossref","unstructured":"Tang, N., Fu, M., & Wu, J. (2025). Minimal interaction seperated tuning: A new paradigm for visual adaptation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52734.2025.02347"},{"key":"2864_CR168","unstructured":"Tang, N., Fu, M., Zhu, K., & Wu, J. (2024). Low-rank attention side-tuning for parameter-efficient fine-tuning. arXiv:2402.04009"},{"key":"2864_CR169","doi-asserted-by":"crossref","unstructured":"Tian, Z., Liu, Y., & Sun, Q. (2025). Meta-learning hyperparameters for parameter efficient fine-tuning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52734.2025.02145"},{"key":"2864_CR170","doi-asserted-by":"publisher","first-page":"57970","DOI":"10.52202\/079017-1848","volume":"37","author":"Z Tian","year":"2024","unstructured":"Tian, Z., Chen, Z., & Sun, Q. (2024). Learning de-biased representations for remote-sensing imagery. Advances in Neural Information Processing Systems (NeurIPS), 37, 57970\u201357992.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2864_CR171","doi-asserted-by":"publisher","first-page":"10078","DOI":"10.52202\/068431-0732","volume":"35","author":"Z Tong","year":"2022","unstructured":"Tong, Z., Song, Y., Wang, J., & Wang, L. (2022). Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Advances in Neural Information Processing Systems (NeurIPS), 35, 10078\u201310093.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2864_CR172","unstructured":"Touvron, H., Cord, M., Douze, M., Massa, F., Sablayrolles, A. & J\u00e9gou, H. (2021). Training data-efficient image transformers & distillation through attention. Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"2864_CR173","doi-asserted-by":"publisher","first-page":"27897","DOI":"10.52202\/075280-1211","volume":"36","author":"Y-Y Tsai","year":"2023","unstructured":"Tsai, Y.-Y., Mao, C., & Yang, J. (2023). Convolutional visual prompt for robust visual perception. Advances in Neural Information Processing Systems(NeurIPS), 36, 27897\u201327921.","journal-title":"Advances in Neural Information Processing Systems(NeurIPS)"},{"key":"2864_CR174","doi-asserted-by":"crossref","unstructured":"Tu, C.-H., Mai, Z., & Chao, W.-L. (2023) Visual query tuning: Towards effective usage of intermediate representations for parameter and memory efficient transfer learning. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52729.2023.00746"},{"key":"2864_CR175","doi-asserted-by":"crossref","unstructured":"Van\u00a0Horn, G., Branson, S., Farrell, R., Haber, S., Barry, J., Ipeirotis, P., Perona, P., & Belongie, S. (2015). Building a bird recognition app and large scale dataset with citizen scientists: The fine print in fine-grained dataset collection. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR.2015.7298658"},{"key":"2864_CR176","doi-asserted-by":"crossref","unstructured":"Veeling, B.S., Linmans, J., Winkens, J., Cohen, T., & Welling, M. (2018). Rotation equivariant cnns for digital pathology. Medical Image Computing and Computer Assisted Intervention (MICCAI),.","DOI":"10.1007\/978-3-030-00934-2_24"},{"key":"2864_CR177","unstructured":"Wah, C., Branson, S., Welinder, P., Perona, P., & Belongie, S. (2011). The caltech-ucsd birds-200-2011 dataset. California Institute of Technology."},{"key":"2864_CR178","doi-asserted-by":"crossref","unstructured":"Wang, H., Chang, J., Zhai, Y., Luo, X., Sun, J., Lin, Z., & Tian, Q. (2024). Lion: Implicit vision prompt tuning. Proceedings of the AAAI Conference on Artificial Intelligence (AAAI).","DOI":"10.1609\/aaai.v38i6.28345"},{"key":"2864_CR179","doi-asserted-by":"crossref","unstructured":"Wang, L., Chen, S., Jiang, L., Pan, S., Cai, R., Yang, S., & Yang, F. (2024). Parameter-efficient fine-tuning in large models: A survey of methodologies. arXiv:2410.19878","DOI":"10.21203\/rs.3.rs-5393239\/v1"},{"key":"2864_CR180","doi-asserted-by":"crossref","unstructured":"Wang, Y., Ding, P., Li, L., Cui, C., Ge, Z., Tong, X., Song, W., Zhao, H., Zhao, W., Hou, P., & Huang, S. (2025). Vla-adapter: An effective paradigm for tiny-scale vision-language-action model. Proceedings of the AAAI Conference on Artificial Intelligence (AAAI),.","DOI":"10.1609\/aaai.v40i22.38931"},{"key":"2864_CR181","unstructured":"Wang, H., Ge, S., Lipton, Z., & Xing, E.P. (2019). Learning robust global representations by penalizing local predictive power. Advances in Neural Information Processing Systems (NeurIPS),."},{"key":"2864_CR182","doi-asserted-by":"crossref","unstructured":"Wang, Y., Shi, B., Zhang, X., Li, J., Liu, Y., Dai, W., Li, C., Xiong, H., & Tian, Q. (2023). Adapting shortcut with normalizing flow: An efficient tuning framework for visual recognition. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52729.2023.01532"},{"key":"2864_CR183","doi-asserted-by":"crossref","unstructured":"Wang, W., Xie, E., Li, X., Fan, D.-P., Song, K., Liang, D., Lu, T., Luo, P., & Shao, L. (2021). Pyramid vision transformer: A versatile backbone for dense prediction without convolutions. Proceedings of the IEEE International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV48922.2021.00061"},{"key":"2864_CR184","doi-asserted-by":"publisher","first-page":"14388","DOI":"10.52202\/068431-1046","volume":"35","author":"Z Wang","year":"2022","unstructured":"Wang, Z., Yu, X., Rao, Y., Zhou, J., & Lu, J. (2022). P2p: Tuning pre-trained image models for point cloud analysis with point-to-pixel prompting. Advances in Neural Information Processing Systems (NeurIPS), 35, 14388\u201314402.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2864_CR185","doi-asserted-by":"crossref","unstructured":"Wei, Y., Zhang, S., Qing, Z., Yuan, H., Liu, Z., Liu, Y., Zhang, Y., Zhou, J., & Shan, H. (2024). Dreamvideo: Composing your dream videos with customized subject and motion. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52733.2024.00625"},{"key":"2864_CR186","unstructured":"Wu, J., Li, X., Wei, C., Wang, H., Yuille, A., Zhou, Y., & Xie, C. (2022). Unleashing the power of visual prompting at the pixel level. Transactions on Machine Learning Research (TMLR)."},{"key":"2864_CR187","unstructured":"Wu, Y., Shi, Y., Wei, J., Sun, C., Yang, Y., & Shen, H.T. (2024). Difflora: Generating personalized low-rank adaptation weights with diffusion. arXiv:2408.06740"},{"key":"2864_CR188","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2025.103547","volume":"102","author":"J Wu","year":"2025","unstructured":"Wu, J., Wang, Z., Hong, M., Ji, W., Fu, H., Xu, Y., Xu, M., & Jin, Y. (2025). Medical sam adapter: Adapting segment anything model for medical image segmentation. Medical Image Analysis, 102, Article 103547.","journal-title":"Medical Image Analysis"},{"key":"2864_CR189","doi-asserted-by":"crossref","unstructured":"Xia, S., Wang, W., Wang, Z., Zhang, Y., Jin, Y., Meng, D., & Hou, R. (2025). Cryptpeft: Efficient and private neural network inference via parameter-efficient fine-tuning. arXiv:2508.12264","DOI":"10.14722\/ndss.2026.231102"},{"key":"2864_CR190","unstructured":"Xia, J., Zhang, C., Zhang, Y., Zhou, C., Wang, Z., Liu, B., & Yin, D. (2025). Dape: Dual-stage parameter-efficient fine-tuning for consistent video editing with diffusion models. arXiv:2505.07057"},{"key":"2864_CR191","doi-asserted-by":"crossref","unstructured":"Xiao, J., Hays, J., Ehinger, K.A., Oliva, A., & Torralba, A. (2010). Sun database: Large-scale scene recognition from abbey to zoo. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"2864_CR192","doi-asserted-by":"crossref","unstructured":"Xiao, A., Xuan, W., Wang, J., Huang, J., Tao, D., Lu, S., & Yokoya, N. (2025). Foundation models for remote sensing and earth observation: A survey. IEEE Geoscience and Remote Sensing Magazine, .","DOI":"10.1109\/MGRS.2025.3576766"},{"key":"2864_CR193","doi-asserted-by":"crossref","unstructured":"Xiao, X., Zhang, Y., Li, X., Wang, T., Wang, X., Wei, Y., Hamm, J., & Xu, M. (2025). Visual instance-aware prompt tuning. In: Proceedings of the ACM Conference on Multimedia (MM).","DOI":"10.1145\/3746027.3754858"},{"key":"2864_CR194","doi-asserted-by":"crossref","unstructured":"Xie, E., Yao, L., Shi, H., Liu, Z., Zhou, D., Liu, Z., Li, J., & Li, Z. (2023). Difffit: Unlocking transferability of large diffusion models via simple parameter-efficient fine-tuning. Proceedings of the IEEE International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV51070.2023.00390"},{"key":"2864_CR195","doi-asserted-by":"crossref","unstructured":"Xie, Z., Zhang, Z., Cao, Y., Lin, Y., Bao, J., Yao, Z., Dai, Q., & Hu, H. (2021). Simmim: a simple framework for masked image modeling. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52688.2022.00943"},{"key":"2864_CR196","doi-asserted-by":"crossref","unstructured":"Xin, Y., Du, J., Wang, Q., Lin, Z., & Yan, K. (2024). Vmt-adapter: Parameter-efficient transfer learning for multi-task dense. Proceedings of the AAAI Conference on Artificial Intelligence (AAAI),.","DOI":"10.1609\/aaai.v38i14.29541"},{"key":"2864_CR197","doi-asserted-by":"crossref","unstructured":"Xin, Y., Du, J., Wang, Q., Yan, K., & Ding, S. (2024). Mmap: Multi-modal alignment prompt for cross-domain multi-task learning. Proceedings of the AAAI Conference on Artificial Intelligence (AAAI),.","DOI":"10.1609\/aaai.v38i14.29540"},{"key":"2864_CR198","doi-asserted-by":"crossref","unstructured":"Xing, Z., Dai, Q., Hu, H., Chen, J., Wu, Z., & Jiang, Y.-G. (2023). Svformer: Semi-supervised video transformer for action recognition. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52729.2023.01804"},{"key":"2864_CR199","doi-asserted-by":"crossref","unstructured":"Xing, Z., Dai, Q., Hu, H., Wu, Z., & Jiang, Y.-G. (2024). Simda: Simple diffusion adapter for efficient video generation. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52733.2024.00748"},{"key":"2864_CR200","doi-asserted-by":"crossref","unstructured":"Xu, X., Xia, C., Wang, Z., Zhao, L., Duan, Y., Zhou, J., & Lu, J. (2024). Memory-based adapters for online 3d scene perception. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52733.2024.02041"},{"key":"2864_CR201","doi-asserted-by":"crossref","unstructured":"Xu, L., Xie, H., Qin, S. J., Tao, X., & Wang, F. L. (2026). Parameter-efficient fine-tuning methods for pretrained language models: A critical review and assessment. IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI),.","DOI":"10.1109\/TPAMI.2026.3657354"},{"key":"2864_CR202","unstructured":"Xu, C., Yang, S., Wang, Y., Wang, Z., Fu, Y., & Xue, X. (2023). Exploring efficient few-shot adaptation for vision transformers. Transactions on Machine Learning Research (TMLR)."},{"key":"2864_CR203","doi-asserted-by":"crossref","unstructured":"Xu, M., Zhang, Z., Wei, F., Hu, H., & Bai, X. (2023). Side adapter network for open-vocabulary semantic segmentation. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52729.2023.00288"},{"key":"2864_CR204","doi-asserted-by":"crossref","unstructured":"Yang, J., Dong, X., Liu, L., Zhang, C., Shen, J., & Yu, D. (2022). Recurring the transformer for video action recognition.  Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52688.2022.01367"},{"key":"2864_CR205","doi-asserted-by":"crossref","unstructured":"Yang, Y., Jiang, P.-T., Hou, Q., Zhang, H., Chen, J., & Li, B. (2024). Multi-task dense prediction via mixture of low-rank experts. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52733.2024.02638"},{"key":"2864_CR206","unstructured":"Yang, T., Zhu, Y., Xie, Y., Zhang, A., Chen, C., & Li, M. (2023). Aim: Adapting image models for efficient video action recognition. Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"2864_CR207","unstructured":"Ye, H., Zhang, J., Liu, S., Han, X., & Yang, W. (2024). Ip-adapter: Text compatible image prompt adapter for text-to-image diffusion models. arXiv:2308.06721"},{"key":"2864_CR208","doi-asserted-by":"crossref","unstructured":"Yin, D., Han, X., Li, B., Feng, H., & Bai, J. (2024). Parameter-efficient is not sufficient: Exploring parameter, memory, and time efficient adapter tuning for dense predictions. Proceedings of the ACM Conference on Multimedia (MM).","DOI":"10.1145\/3664647.3680940"},{"key":"2864_CR209","unstructured":"Yin, D., Hu, L., Li, B., & Zhang, Y. (2023). Adapter is all you need for tuning visual tasks. arXiv:2311.15010"},{"key":"2864_CR210","doi-asserted-by":"crossref","unstructured":"Yin, D., Hu, L., Li, B., Zhang, Y., & Yang, X. (2025). 5%$$>$$ 100%: Breaking performance shackles of full fine-tuning on visual recognition tasks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52734.2025.01869"},{"key":"2864_CR211","doi-asserted-by":"crossref","unstructured":"Yin, D., Yang, Y., Wang, Z., Yu, H., Wei, K., & Sun, X. (2023). 1% vs 100%: Parameter-efficient low rank adapter for dense predictions. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52729.2023.01926"},{"key":"2864_CR212","doi-asserted-by":"crossref","unstructured":"Yin, D., Zhao, T.-F., Fan, D.-P., Li, S., Du, B., Sun, X., & Hu, S.-M. (2025). Remote sensing tuning: A survey. Computational Visual Media.","DOI":"10.26599\/CVM.2025.9450490"},{"key":"2864_CR213","unstructured":"Yu, B.X., Chang, J., Liu, L., Tian, Q., & Chen, C.W. (2022). Towards a unified view on visual parameter-efficient transfer learning. arXiv:2210.00788"},{"key":"2864_CR214","doi-asserted-by":"crossref","unstructured":"Yu, H., Yin, W., Bi, H., Li, C., Feng, Y., Diao, W., & Sun, X. (2025). Efficient side-tuning for remote sensing: a low-memory fine-tuning framework. IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing.","DOI":"10.1109\/JSTARS.2025.3563641"},{"key":"2864_CR215","doi-asserted-by":"crossref","unstructured":"Yuan, L., Chen, Y., Wang, T., Yu, W., Shi, Y., Jiang, Z.-H., Tay, F.E., Feng, J., & Yan, S. (2021). Tokens-to-token vit: Training vision transformers from scratch on imagenet. Proceedings of the IEEE International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV48922.2021.00060"},{"key":"2864_CR216","doi-asserted-by":"crossref","unstructured":"Zaken, E.B., Goldberg, Y., & Ravfogel, S. (2022). Bitfit: Simple parameter-efficient fine-tuning for transformer-based masked language-models. Proceedings of the Annual Meeting of the Association for Computational Linguistics (ACL).","DOI":"10.18653\/v1\/2022.acl-short.1"},{"key":"2864_CR217","doi-asserted-by":"publisher","first-page":"5552","DOI":"10.52202\/079017-0180","volume":"37","author":"R Zeng","year":"2024","unstructured":"Zeng, R., Han, C., Wang, Q., Wu, C., Geng, T., Huang, L., Wu, Y. N., & Liu, D. (2024). Visual fourier prompt tuning. Advances in Neural Information Processing Systems (NeurIPS), 37, 5552\u20135585.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2864_CR218","doi-asserted-by":"crossref","unstructured":"Zha, Y., Wang, J., Dai, T., Chen, B., Wang, Z., & Xia, S.-T. (2023). Instance-aware dynamic prompt tuning for pre-trained point cloud models. Proceedings of the IEEE International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV51070.2023.01302"},{"key":"2864_CR219","doi-asserted-by":"crossref","unstructured":"Zha, Y., Wang, Y., Guo, H., Wang, J., Dai, T., Chen, B., Ouyang, Z., Yuerong, X., Chen, K., & Xia, S.-T. (2025). Pma: Towards parameter-efficient point cloud understanding via point mamba adapter. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52734.2025.01582"},{"key":"2864_CR220","unstructured":"Zhai, X., Puigcerver, J., Kolesnikov, A., Ruyssen, P., Riquelme, C., Lucic, M., Djolonga, J., Pinto, A.S., Neumann, M., Dosovitskiy, A., & Beyer, L. (2019). A large-scale study of representation learning with the visual task adaptation benchmark. arXiv:1910.04867"},{"key":"2864_CR221","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Chen, H., Zhang, X., Chu, X., & Song, L. (2025). Dyn-adapter: Towards disentangled representation for efficient visual recognition. Proceedings of the European Conference on Computer Vision (ECCV),.","DOI":"10.1007\/978-3-031-73209-6_25"},{"key":"2864_CR222","doi-asserted-by":"crossref","unstructured":"Zhang, L., Rao, A., & Agrawala, M. (2023). Adding conditional control to text-to-image diffusion models. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"2864_CR223","doi-asserted-by":"crossref","unstructured":"Zhang, J.O., Sax, A., Zamir, A., Guibas, L., & Malik, J. (2020). Side-tuning: a baseline for network adaptation via additive side networks. Proceedings of the European Conference on Computer Vision (ECCV).","DOI":"10.1007\/978-3-030-58580-8_41"},{"key":"2864_CR224","unstructured":"Zhang, D., Yan, R., Dong, P., & Cheng, K.-T. (2025). Memory efficient transformer adapter for dense predictions. arXiv:2502.01962"},{"key":"2864_CR225","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Zhang, Q., Gao, Z., Zhang, R., Shutova, E., Zhou, S., & Zhang, S. (2024). Gradient-based parameter selection for efficient fine-tuning. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52733.2024.02699"},{"key":"2864_CR226","unstructured":"Zhang, Q., Zou, B., An, R., Liu, J., & Zhang, S. (2023). Mosa: Mixture of sparse adapters for visual efficient tuning. arXiv:2312.02923"},{"issue":"9","key":"2864_CR227","doi-asserted-by":"publisher","first-page":"2363","DOI":"10.3390\/rs15092363","volume":"15","author":"Z Zhang","year":"2023","unstructured":"Zhang, Z., Liu, F., Liu, C., Tian, Q., & Qu, H. (2023). Actnet: A dual-attention adapter with a cnn-transformer network for the semantic segmentation of remote sensing imagery. Remote Sensing, 15(9), 2363.","journal-title":"Remote Sensing"},{"key":"2864_CR228","doi-asserted-by":"publisher","first-page":"4971","DOI":"10.52202\/068431-0359","volume":"35","author":"B Zhang","year":"2022","unstructured":"Zhang, B., Tian, Z., Tang, Q., Chu, X., Wei, X., & Shen, C. (2022). Segvit: Semantic segmentation with plain vision transformers. Advances in Neural Information Processing Systems (NeurIPS), 35, 4971\u20134982.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"issue":"7","key":"2864_CR229","doi-asserted-by":"publisher","first-page":"5268","DOI":"10.1109\/TPAMI.2024.3435939","volume":"47","author":"Y Zhang","year":"2024","unstructured":"Zhang, Y., Zhou, K., & Liu, Z. (2024). Neural prompt search. IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 47(7), 5268\u20135280.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI)"},{"key":"2864_CR230","doi-asserted-by":"publisher","first-page":"64","DOI":"10.1016\/j.isprsjprs.2025.01.020","volume":"221","author":"Y Zhan","year":"2025","unstructured":"Zhan, Y., Xiong, Z., & Yuan, Y. (2025). Skyeyegpt: Unifying remote sensing vision-language tasks via instruction tuning with large language model. ISPRS Journal of Photogrammetry and Remote Sensing, 221, 64\u201377.","journal-title":"ISPRS Journal of Photogrammetry and Remote Sensing"},{"key":"2864_CR231","doi-asserted-by":"crossref","unstructured":"Zhao, D., Li, J., Wang, S., Wu, M., Zang, Q., Sebe, N., & Zhong, Z. (2025). Fishertune: Fisher-guided robust tuning of vision foundation models for domain generalized segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52734.2025.01401"},{"key":"2864_CR232","doi-asserted-by":"crossref","unstructured":"Zhao, H.H., Wang, P., Zhao, Y., Luo, H., Wang, F., & Shou, M.Z. (2023). Sct: A simple baseline for parameter-efficient fine-tuning via salient channels. International Journal of Computer Vision (IJCV).","DOI":"10.1007\/s11263-023-01918-3"},{"key":"2864_CR233","first-page":"11127","volume":"36","author":"S Zhao","year":"2024","unstructured":"Zhao, S., Chen, D., Chen, Y.-C., Bao, J., Hao, S., Yuan, L., & Wong, K.-Y.K. (2024). Uni-controlnet: All-in-one control to text-to-image diffusion models. Advances in Neural Information Processing Systems (NeurIPS), 36, 11127\u201311150.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2864_CR234","doi-asserted-by":"publisher","first-page":"114765","DOI":"10.52202\/079017-3643","volume":"37","author":"W Zhao","year":"2024","unstructured":"Zhao, W., Tang, J., Han, Y., Song, Y., Wang, K., Huang, G., Wang, F., & You, Y. (2024). Dynamic tuning towards parameter and inference efficiency for vit adaptation. Advances in Neural Information Processing Systems (NeurIPS), 37, 114765\u2013114796.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2864_CR235","doi-asserted-by":"publisher","first-page":"81130","DOI":"10.52202\/079017-2579","volume":"37","author":"H Zhong","year":"2024","unstructured":"Zhong, H., Chen, J., Zhang, Y., Huang, D., & Wang, Y. (2024). Transforming vision transformer: Towards efficient multi-task asynchronous learner. Advances in Neural Information Processing Systems (NeurIPS), 37, 81130\u201381156.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2864_CR236","doi-asserted-by":"crossref","unstructured":"Zhou, X., Liang, D., Xu, W., Zhu, X., Xu, Y., Zou, Z., & Bai, X. (2024). Dynamic adapter meets prompt tuning: Parameter-efficient transfer learning for point cloud analysis. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52733.2024.01393"},{"key":"2864_CR237","doi-asserted-by":"crossref","unstructured":"Zhou, B., Zhao, H., Puig, X., Fidler, S., Barriuso, A., & Torralba, A. (2017). Scene parsing through ade20k dataset. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR.2017.544"},{"key":"2864_CR238","doi-asserted-by":"crossref","unstructured":"Zhou, D.-W., Cai, Z.-W., Ye, H.-J., Zhan, D.-C., & Liu, Z. (2025). Revisiting class-incremental learning with pre-trained models: Generalizability and adaptivity are all you need. International Journal of Computer Vision (IJCV),133(3), 1012\u20131032.","DOI":"10.1007\/s11263-024-02218-0"},{"issue":"3","key":"2864_CR239","doi-asserted-by":"publisher","first-page":"302","DOI":"10.1007\/s11263-018-1140-0","volume":"127","author":"B Zhou","year":"2019","unstructured":"Zhou, B., Zhao, H., Puig, X., Xiao, T., Fidler, S., Barriuso, A., & Torralba, A. (2019). Semantic understanding of scenes through the ade20k dataset. International Journal of Computer Vision (IJCV), 127(3), 302\u2013321.","journal-title":"International Journal of Computer Vision (IJCV)"},{"key":"2864_CR240","doi-asserted-by":"crossref","unstructured":"Zhu, J., Lai, S., Chen, X., Wang, D., & Lu, H. (2023). Visual prompt multi-modal tracking. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52729.2023.00918"},{"key":"2864_CR241","doi-asserted-by":"crossref","unstructured":"Zhu, M., Liu, G., & Wei, Z. (2024). Vl-mpft: Multitask parameter-efficient fine-tuning for visual-language pre-trained models via task-adaptive masking. Chinese Conference on Pattern Recognition and Computer Vision (PRCV),.","DOI":"10.1007\/978-981-97-8620-6_26"},{"key":"2864_CR242","doi-asserted-by":"crossref","unstructured":"Zhu, H., Zhang, Y., Dong, J., & Koniusz, P. (2025). Bilora: Almost-orthogonal parameter spaces for continual learning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR),.","DOI":"10.1109\/CVPR52734.2025.02385"},{"key":"2864_CR243","doi-asserted-by":"crossref","unstructured":"Zhu, H., Zhang, F., Qin, R., Pan, T., Yong, J., & Wang, B. (2024). Semantic hierarchical prompt tuning for parameter-efficient fine-tuning. arXiv:2412.16956","DOI":"10.1109\/ICASSP49660.2025.10889615"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02864-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-026-02864-6","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02864-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T17:02:30Z","timestamp":1784566950000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-026-02864-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":243,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["2864"],"URL":"https:\/\/doi.org\/10.1007\/s11263-026-02864-6","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"8 December 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 April 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 June 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"304"}}