{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T15:40:07Z","timestamp":1780501207367,"version":"3.54.1"},"reference-count":85,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2023,10,16]],"date-time":"2023-10-16T00:00:00Z","timestamp":1697414400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,10,16]],"date-time":"2023-10-16T00:00:00Z","timestamp":1697414400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2024,3]]},"DOI":"10.1007\/s11263-023-01918-3","type":"journal-article","created":{"date-parts":[[2023,10,16]],"date-time":"2023-10-16T18:02:56Z","timestamp":1697479376000},"page":"731-749","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":18,"title":["SCT: A Simple Baseline for Parameter-Efficient Fine-Tuning via Salient Channels"],"prefix":"10.1007","volume":"132","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8047-4465","authenticated-orcid":false,"given":"Henry Hengyuan","family":"Zhao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pichao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuyang","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hao","family":"Luo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mike Zheng","family":"Shou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,10,16]]},"reference":[{"key":"1918_CR1","unstructured":"Ali, A., Touvron, H., Caron, M., Bojanowski, P., Douze, M., Joulin, A., Laptev, I., Neverova, N., Synnaeve, G., Verbeek, J., et al. (2021). Xcit: Cross-covariance image transformers. In NeurIPS."},{"key":"1918_CR2","unstructured":"Bar, A., Gandelsman, Y., Darrell, T., Globerson, A., & Efros, A.A. (2022). Visual prompting via image inpainting. arXiv preprint arXiv:2209.00647."},{"key":"1918_CR3","unstructured":"Beattie, C., Leibo, J.Z., Teplyashin, D., Ward, T., Wainwright, M., K\u00fcttler, H., Lefrancq, A., Green, S., Vald\u00e9s, V., Sadik, A., et\u00a0al. (2016). Deepmind lab. arXiv preprint arXiv:1612.03801."},{"key":"1918_CR4","doi-asserted-by":"crossref","unstructured":"Bossard, L., Guillaumin, M., & Gool, L.V. (2014). Food-101\u2013mining discriminative components with random forests. In European conference on computer vision (ECCV), Springer, pp 446\u2013461.","DOI":"10.1007\/978-3-319-10599-4_29"},{"key":"1918_CR5","unstructured":"Cai, H., Gan, C., Zhu, L., & Han, S. (2020). Tinytl: Reduce memory, not parameters for efficient on-device learning. In NeurIPS."},{"key":"1918_CR6","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., & Zagoruyko, S. (2020). End-to-end object detection with transformers. In European conference on computer vision, Springer, pp 213\u2013229.","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"1918_CR7","doi-asserted-by":"crossref","unstructured":"Chen, C.F.R., Fan, Q., & Panda, R. (2021a). Crossvit: Cross-attention multi-scale vision transformer for image classification. In Proceedings of the IEEE\/CVF international conference on computer vision, pp 357\u2013366.","DOI":"10.1109\/ICCV48922.2021.00041"},{"key":"1918_CR8","unstructured":"Chen, H., Tao, R., Zhang, H., Wang, Y., Ye, W., Wang, J., Hu, G., & Savvides, M. (2022a). Conv-adapter: Exploring parameter efficient transfer learning for convnets. arXiv preprint arXiv:2208.07463."},{"key":"1918_CR9","unstructured":"Chen, S., Ge, C., Tong, Z., Wang, J., Song, Y., Wang, J., & Luo, P. (2022b). Adaptformer: Adapting vision transformers for scalable visual recognition. arXiv preprint arXiv:2205.13535."},{"key":"1918_CR10","doi-asserted-by":"crossref","unstructured":"Chen, X., Xie, S., & He, K. (2021b). An empirical study of training self-supervised vision transformers. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 9640\u20139649.","DOI":"10.1109\/ICCV48922.2021.00950"},{"issue":"10","key":"1918_CR11","doi-asserted-by":"publisher","first-page":"1865","DOI":"10.1109\/JPROC.2017.2675998","volume":"105","author":"G Cheng","year":"2017","unstructured":"Cheng, G., Han, J., & Lu, X. (2017). Remote sensing image scene classification: Benchmark and state of the art. Proceedings of the IEEE, 105(10), 1865\u20131883.","journal-title":"Proceedings of the IEEE"},{"key":"1918_CR12","doi-asserted-by":"crossref","unstructured":"Cimpoi, M., Maji, S., Kokkinos, I., Mohamed, S., & Vedaldi, A. (2014). Describing textures in the wild. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR.2014.461"},{"key":"1918_CR13","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., & Fei-Fei, L. (2009). Imagenet: A large-scale hierarchical image database. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), IEEE, pp 248\u2013255.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"1918_CR14","doi-asserted-by":"crossref","unstructured":"Dong, X., Bao, J., Chen, D., Zhang, W., Yu, N., Yuan, L., Chen, D., & Guo, B. (2022). Cswin transformer: A general vision transformer backbone with cross-shaped windows. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12124\u201312134.","DOI":"10.1109\/CVPR52688.2022.01181"},{"key":"1918_CR15","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., et\u00a0al. (2020). An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929."},{"key":"1918_CR16","doi-asserted-by":"crossref","unstructured":"d\u2019Ascoli, S., Touvron, H., Leavitt, M.L., Morcos, A.S., Biroli, G., & Sagun, L. (2021). Convit: Improving vision transformers with soft convolutional inductive biases. In International Conference on Machine Learning, PMLR, pp 2286\u20132296.","DOI":"10.1088\/1742-5468\/ac9830"},{"key":"1918_CR17","doi-asserted-by":"crossref","unstructured":"Fan, H., Xiong, B., Mangalam, K., Li, Y., Yan, Z., Malik, J., & Feichtenhofer, C. (2021). Multiscale vision transformers. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 6824\u20136835.","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"1918_CR18","doi-asserted-by":"crossref","unstructured":"Fei-Fei, L., Fergus, R., & Perona, P. (2004). Learning generative visual models from few training examples: An incremental bayesian approach tested on 101 object categories. InConference on computer vision and pattern recognition workshop, IEEE, pp 178\u2013178.","DOI":"10.1109\/CVPR.2004.383"},{"issue":"11","key":"1918_CR19","doi-asserted-by":"publisher","first-page":"1231","DOI":"10.1177\/0278364913491297","volume":"32","author":"A Geiger","year":"2013","unstructured":"Geiger, A., Lenz, P., Stiller, C., & Urtasun, R. (2013). Vision meets robotics: The kitti dataset. The International Journal of Robotics Research, 32(11), 1231\u20131237.","journal-title":"The International Journal of Robotics Research"},{"key":"1918_CR20","unstructured":"Han, K., Xiao, A., Wu, E., Guo, J., Xu, C., & Wang, Y. (2021). Transformer in transformer. In NeurIPS."},{"key":"1918_CR21","unstructured":"Han, S., Pool, J., Tran, J., & Dally, W. (2015). Learning both weights and connections for efficient neural network. Advances in neural information processing systems 28."},{"key":"1918_CR22","doi-asserted-by":"crossref","unstructured":"He, Y., Kang, G., Dong, X., Fu, Y., & Yang, Y. (2018). Soft filter pruning for accelerating deep convolutional neural networks. In IJCAI International Joint Conference on Artificial Intelligence.","DOI":"10.24963\/ijcai.2018\/309"},{"key":"1918_CR23","doi-asserted-by":"publisher","unstructured":"Helber, P., Bischke, B., Dengel, A., & Borth, D. (2019). Eurosat: A novel dataset and deep learning benchmark for land use and land cover classification. IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing, 12(7), 2217\u20132226. https:\/\/doi.org\/10.1109\/JSTARS.2019.2918242","DOI":"10.1109\/JSTARS.2019.2918242"},{"key":"1918_CR24","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Basart, S., Mu, N., Kadavath, S., Wang, F., Dorundo, E., Desai, R., Zhu, T., Parajuli, S., Guo, M., et\u00a0al. (2021a). The many faces of robustness: A critical analysis of out-of-distribution generalization. In Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp 8340\u20138349.","DOI":"10.1109\/ICCV48922.2021.00823"},{"key":"1918_CR25","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Zhao, K., Basart, S., Steinhardt, J., & Song, D. (2021b). Natural adversarial examples. In: Proceedings of the IEEE\/CVF Conference on computer vision and pattern recognition (CVPR), pp 15262\u201315271.","DOI":"10.1109\/CVPR46437.2021.01501"},{"key":"1918_CR26","unstructured":"Houlsby, N., Giurgiu, A., Jastrzebski, S., Morrone, B., De\u00a0Laroussilhe, Q., Gesmundo, A., Attariyan, M., & Gelly, S. (2019). Parameter-efficient transfer learning for nlp. In: International conference on machine learning (ICML), PMLR, pp 2790\u20132799."},{"key":"1918_CR27","unstructured":"Hu, E.J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., & Chen, W. (2021). Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685."},{"key":"1918_CR28","doi-asserted-by":"crossref","unstructured":"Jia, M., Wu, Z., Reiter, A., Cardie, C., Belongie, S., & Lim, S.N. (2021). Exploring visual engagement signals for representation learning. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 4206\u20134217.","DOI":"10.1109\/ICCV48922.2021.00417"},{"key":"1918_CR29","doi-asserted-by":"crossref","unstructured":"Jia, M., Tang, L., Chen, B.C., Cardie, C., Belongie, S., Hariharan, B., & Lim, S.N. (2022). Visual prompt tuning. In ECCV.","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"1918_CR30","unstructured":"Jie, S., & Deng, Z.H. (2022). Convolutional bypasses are better vision transformer adapters. arXiv preprint arXiv:2207.07039."},{"key":"1918_CR31","doi-asserted-by":"crossref","unstructured":"Johnson, J., Hariharan, B., Van Der\u00a0Maaten, L., Fei-Fei, L., Lawrence\u00a0Zitnick, C., & Girshick, R. (2017). Clevr: A diagnostic dataset for compositional language and elementary visual reasoning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 2901\u20132910.","DOI":"10.1109\/CVPR.2017.215"},{"key":"1918_CR32","unstructured":"Kaggle & EyePacs (2015). Kaggle diabetic retinopathy detection https:\/\/www.kaggle.com\/c\/diabetic-retinopathy-detection\/data."},{"key":"1918_CR33","doi-asserted-by":"crossref","unstructured":"Krause, J., Stark, M., Deng, J., & Fei-Fei, L. (2013). 3d object representations for fine-grained categorization. In Proceedings of the IEEE international conference on computer vision workshops, pp 554\u2013561.","DOI":"10.1109\/ICCVW.2013.77"},{"key":"1918_CR34","unstructured":"Krizhevsky, A., Hinton, G., et\u00a0al. (2009). Learning multiple layers of features from tiny images."},{"key":"1918_CR35","doi-asserted-by":"crossref","unstructured":"LeCun, Y., Huang, F.J., & Bottou, L. (2004). Learning methods for generic object recognition with invariance to pose and lighting. In Proceedings of the IEEE\/CVF Conference on computer vision and pattern recognition (CVPR), IEEE, vol\u00a02, pp II\u2013104.","DOI":"10.1109\/CVPR.2004.1315150"},{"key":"1918_CR36","unstructured":"Li, H., Kadav, A., Durdanovic, I., Samet, H., & Graf, H.P. (2016). Pruning filters for efficient convnets. arXiv preprint arXiv:1608.08710."},{"key":"1918_CR37","unstructured":"Li, H., Kadav, A., Durdanovic, I., Samet, H., & Graf, H.P. (2017). Pruning filters for efficient convnets. In: International conference on learning representations, https:\/\/openreview.net\/forum?id=rJqFGTslg."},{"key":"1918_CR38","unstructured":"Li, Y., Xie, S., Chen, X., Dollar, P., He, K., & Girshick, R. (2021). Benchmarking detection transfer learning with vision transformers. arXiv preprint arXiv:2111.11429."},{"key":"1918_CR39","unstructured":"Lian, D., Zhou, D., Feng, J., & Wang, X. (2022). Scaling & shifting your features: A new baseline for efficient model tuning. In Advances in neural information processing systems (NeurIPS)."},{"key":"1918_CR40","unstructured":"Liao, N., Shi, B., Cao, M., Zhang, X., Tian, Q., & Yan, J. (2023). Rethinking visual prompt learning as masked visual token modeling. arXiv preprint arXiv:2303.04998."},{"key":"1918_CR41","unstructured":"Liu, L., Yu, B.X., Chang, J., Tian, Q., & Chen, C.W. (2022). Prompt-matched semantic segmentation. arXiv preprint arXiv:2208.10159."},{"key":"1918_CR42","unstructured":"Liu, Z., Sun, M., Zhou, T., Huang, G., & Darrell, T. (2018). Rethinking the value of network pruning. In: International conference on learning representations."},{"key":"1918_CR43","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., & Guo, B. (2021). Swin transformer: Hierarchical vision transformer using shifted windows. In Proceedings of the IEEE\/CVF international conference on computer vision, pp 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"1918_CR44","unstructured":"Loshchilov, I., & Hutter, F. (2017). Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101."},{"key":"1918_CR45","doi-asserted-by":"crossref","unstructured":"Luo, J.H., Wu, J., & Lin, W. (2017). Thinet: A filter level pruning method for deep neural network compression. In: Proceedings of the IEEE international conference on computer vision, pp 5058\u20135066.","DOI":"10.1109\/ICCV.2017.541"},{"key":"1918_CR46","unstructured":"Luo, X., Xu, J., & Xu, Z. (2022). Channel importance matters in few-shot image classification. In: International conference on machine learning, PMLR, pp 14542\u201314559."},{"key":"1918_CR47","doi-asserted-by":"crossref","unstructured":"Mahajan, D., Girshick, R., Ramanathan, V., He, K., Paluri, M., Li, Y., Bharambe, A., & Van Der\u00a0Maaten, L. (2018). Exploring the limits of weakly supervised pretraining. In: Proceedings of the European conference on computer vision (ECCV), pp 181\u2013196.","DOI":"10.1007\/978-3-030-01216-8_12"},{"key":"1918_CR48","unstructured":"Maji, S., Rahtu, E., Kannala, J., Blaschko, M., & Vedaldi, A. (2013). Fine-grained visual classification of aircraft. arXiv preprint arXiv:1306.5151."},{"key":"1918_CR49","unstructured":"Manli, S., Weili, N., De-An, H., Zhiding, Y., Tom, G., Anima, A., & Chaowei, X. (2022). Test-time prompt tuning for zero-shot generalization in vision-language models. In NeurIPS."},{"key":"1918_CR50","unstructured":"Matthey, L., Higgins, I., Hassabis, D., & Lerchner, A. (2017). dsprites: Disentanglement testing sprites dataset."},{"key":"1918_CR51","unstructured":"Netzer, Y., Wang, T., Coates, A., Bissacco, A., Wu, B., & Ng, A.Y. (2011). Reading digits in natural images with unsupervised feature learning."},{"key":"1918_CR52","unstructured":"Nie, X., Ni, B., Chang, J., Meng, G., Huo, C., Zhang, Z., Xiang, S., Tian, Q., & Pan, C. (2022). Pro-tuning: Unified prompt tuning for vision tasks. arXiv preprint arXiv:2207.14381."},{"key":"1918_CR53","doi-asserted-by":"crossref","unstructured":"Nilsback, M.E., & Zisserman, A. (2006). A visual vocabulary for flower classification. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), IEEE, vol\u00a02, pp 1447\u20131454.","DOI":"10.1109\/CVPR.2006.42"},{"key":"1918_CR54","first-page":"26462","volume":"35","author":"J Pan","year":"2022","unstructured":"Pan, J., Lin, Z., Zhu, X., Shao, J., & Li, H. (2022). St-adapter: Parameter-efficient image-to-video transfer learning. Advances in Neural Information Processing Systems, 35, 26462\u201326477.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"1918_CR55","doi-asserted-by":"crossref","unstructured":"Parkhi, O.M., Vedaldi, A., Zisserman, A., & Jawahar, C. (2012). Cats and dogs. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), IEEE, pp 3498\u20133505.","DOI":"10.1109\/CVPR.2012.6248092"},{"key":"1918_CR56","unstructured":"Rao, Y., Zhao, W., Liu, B., Lu, J., Zhou, J., & Hsieh, C.J. (2021). Dynamicvit: Efficient vision transformers with dynamic token sparsification. In NeurIPS."},{"key":"1918_CR57","unstructured":"Recht, B., Roelofs, R., Schmidt, L., & Shankar, V. (2019). Do imagenet classifiers generalize to imagenet? In: International conference on machine learning (ICML), PMLR, pp 5389\u20135400."},{"key":"1918_CR58","doi-asserted-by":"crossref","unstructured":"Strudel, R., Garcia, R., Laptev, I., & Schmid, C. (2021). Segmenter: Transformer for semantic segmentation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 7262\u20137272.","DOI":"10.1109\/ICCV48922.2021.00717"},{"key":"1918_CR59","doi-asserted-by":"crossref","unstructured":"Sung, Y.L., Cho, J., & Bansal, M. (2022). Vl-adapter: Parameter-efficient transfer learning for vision-and-language tasks. In: Proceedings of the IEEE\/CVF Conference on computer vision and pattern recognition, pp 5227\u20135237.","DOI":"10.1109\/CVPR52688.2022.00516"},{"key":"1918_CR60","doi-asserted-by":"crossref","unstructured":"Touvron, H., Cord, M., Sablayrolles, A., Synnaeve, G., & J\u00e9gou, H. (2021). Going deeper with image transformers. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 32\u201342.","DOI":"10.1109\/ICCV48922.2021.00010"},{"key":"1918_CR61","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., & Polosukhin, I. (2017). Attention is all you need. In NeurIPS."},{"key":"1918_CR62","doi-asserted-by":"crossref","unstructured":"Veeling, B.S., Linmans, J., Winkens, J., Cohen, T., & Welling, M. (2018). Rotation equivariant cnns for digital pathology. In International Conference on Medical image computing and computer-assisted intervention, Springer, pp 210\u2013218.","DOI":"10.1007\/978-3-030-00934-2_24"},{"key":"1918_CR63","unstructured":"Wang, H., Ge, S., Lipton, Z., & Xing, E.P. (2019). Learning robust global representations by penalizing local predictive power. In NeurIPS."},{"key":"1918_CR64","doi-asserted-by":"crossref","unstructured":"Wang, P., Wang, X., Wang, F., Lin, M., Chang, S., Xie, W., Li, H., & Jin, R. (2021). Kvt: k-nn attention for boosting vision transformers. arXiv preprint arXiv:2106.00515.","DOI":"10.1007\/978-3-031-20053-3_17"},{"key":"1918_CR65","unstructured":"Wang, S., Chang, J., Wang, Z., Li, H., Ouyang, W., & Tian, Q. (2022). Fine-grained retrieval prompt tuning. arXiv preprint arXiv:2207.14465."},{"key":"1918_CR66","doi-asserted-by":"crossref","unstructured":"Xiao, J., Hays, J., Ehinger, K.A., Oliva, A., & Torralba, A. (2010). Sun database: Large-scale scene recognition from abbey to zoo. In 2010 IEEE computer society conference on computer vision and pattern recognition, IEEE, pp 3485\u20133492.","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"1918_CR67","unstructured":"Xing, Y., Wu, Q., Cheng, D., Zhang, S., Liang, G., & Zhang, Y. (2022). Class-aware visual prompt tuning for vision-language pre-trained model. arXiv preprint arXiv:2208.08340."},{"key":"1918_CR68","doi-asserted-by":"crossref","unstructured":"Xu, R., Luo, F., Zhang, Z., Tan, C., Chang, B., Huang, S., & Huang, F. (2021). Raise a child in large language model: Towards effective and generalizable fine-tuning. In Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing (EMNLP), Association for Computational Linguistics.","DOI":"10.18653\/v1\/2021.emnlp-main.749"},{"key":"1918_CR69","unstructured":"Yang, J., Zhou, K., Li, Y., & Liu, Z. (2021). Generalized out-of-distribution detection: A survey. arXiv preprint arXiv:2110.11334."},{"key":"1918_CR70","unstructured":"Yang, J., Wang, P., Zou, D., Zhou, Z., Ding, K., Peng, W., Wang, H., Chen, G., Li, B., Sun, Y., et\u00a0al. (2022a). Openood: Benchmarking generalized out-of-distribution detection. arXiv preprint arXiv:2210.07242."},{"key":"1918_CR71","unstructured":"Yang, J., Zhou, K., & Liu, Z. (2022b). Full-spectrum out-of-distribution detection. arXiv preprint arXiv:2204.05306."},{"key":"1918_CR72","doi-asserted-by":"crossref","unstructured":"Yuan, L., Chen, Y., Wang, T., Yu, W., Shi, Y., Jiang, Z.H., Tay, F.E., Feng, J., & Yan, S. (2021). Tokens-to-token vit: Training vision transformers from scratch on imagenet. In Proceedings of the IEEE\/CVF International conference on computer vision, pp 558\u2013567.","DOI":"10.1109\/ICCV48922.2021.00060"},{"issue":"5","key":"1918_CR73","first-page":"6575","volume":"45","author":"L Yuan","year":"2022","unstructured":"Yuan, L., Hou, Q., Jiang, Z., Feng, J., & Yan, S. (2022). Volo: Vision outlooker for visual recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(5), 6575\u20136586.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1918_CR74","unstructured":"Zang, Y., Li, W., Zhou, K., Huang, C., & Loy, C.C. (2022). Unified vision and language prompt learning. arXiv preprint arXiv:2210.07225."},{"key":"1918_CR75","unstructured":"Zhai, X., Puigcerver, J., Kolesnikov, A., Ruyssen, P., Riquelme, C., Lucic, M., Djolonga, J., Pinto, A.S., Neumann, M., Dosovitskiy, A., et\u00a0al. (2019). A large-scale study of representation learning with the visual task adaptation benchmark. arXiv preprint arXiv:1910.04867."},{"key":"1918_CR76","unstructured":"Zhang, B., Jin, X., Gong, W., Xu, K., Zhang, Z., Wang, P., Shen, X., & Feng, J. (2023a). Multimodal video adapter for parameter efficient video text retrieval. arXiv preprint arXiv:2301.07868."},{"key":"1918_CR77","unstructured":"Zhang, Y., Zhou, K., & Liu, Z. (2022). Neural prompt search. arXiv preprint arXiv:2206.04673."},{"key":"1918_CR78","unstructured":"Zhang, Y., Zhou, K., & Liu, Z. (2023b). What makes good examples for visual in-context learning?."},{"key":"1918_CR79","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Zhong, Z., Zhao, N., Sebe, N., & Lee, G.H. (2022). Style-hallucinated dual consistency learning for domain generalized semantic segmentation. In ECCV.","DOI":"10.1007\/s11263-023-01911-w"},{"key":"1918_CR80","unstructured":"Zheng, Z., Yue, X., Wang, K., & You, Y. (2022). Prompt vision transformer for domain generalization. arXiv preprint arXiv:2208.08914."},{"key":"1918_CR81","unstructured":"Zhou, J., Wang, P., Wang, F., Liu, Q., Li, H., & Jin, R. (2021a). Elsa: Enhanced local self-attention for vision transformer. arXiv preprint arXiv:2112.12786."},{"key":"1918_CR82","unstructured":"Zhou, K., Yang, Y., Qiao, Y., & Xiang, T. (2021b). Domain generalization with mixstyle. In ICLR."},{"key":"1918_CR83","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C.C., & Liu, Z. (2022a). Conditional prompt learning for vision-language models. In: IEEE\/CVF Conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR52688.2022.01631"},{"issue":"9","key":"1918_CR84","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C. C., & Liu, Z. (2022). Learning to prompt for vision-language models. International Journal of Computer Vision (IJCV), 130(9), 2337\u20132348.","journal-title":"International Journal of Computer Vision (IJCV)"},{"key":"1918_CR85","unstructured":"Zhou, K., Zhang, Y., Zang, Y., Yang, J., Loy, C.C., & Liu, Z. (2022c). On-device domain generalization. arXiv preprint arXiv:2209.07521."}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-023-01918-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-023-01918-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-023-01918-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,16]],"date-time":"2024-02-16T19:14:09Z","timestamp":1708110849000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-023-01918-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,16]]},"references-count":85,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,3]]}},"alternative-id":["1918"],"URL":"https:\/\/doi.org\/10.1007\/s11263-023-01918-3","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,10,16]]},"assertion":[{"value":"30 March 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 September 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 October 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"We declare that there are no conflicts of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}