{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T15:19:46Z","timestamp":1772119186033,"version":"3.50.1"},"reference-count":43,"publisher":"Springer Science and Business Media LLC","issue":"13","license":[{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1007\/s00371-025-04080-8","type":"journal-article","created":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T14:38:27Z","timestamp":1751294307000},"page":"10977-10986","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Two-stage fine-tuning CLIP by introducing structure knowledge for few-shot classification"],"prefix":"10.1007","volume":"41","author":[{"given":"Zhe","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiang-Gui","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junbao","family":"Zhuo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huimin","family":"Ma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,6,30]]},"reference":[{"key":"4080_CR1","doi-asserted-by":"crossref","unstructured":"Renavitasari I.R.D., Supianto A.A.: Educational game for training spatial ability using tangram tuzzle. In: International Conference on Sustainable Information Engineering and Technology, pp. 174\u2013179 (2018)","DOI":"10.1109\/SIET.2018.8693164"},{"key":"4080_CR2","doi-asserted-by":"publisher","first-page":"617","DOI":"10.1146\/annurev.psych.59.103006.093639","volume":"59","author":"LW Barsalou","year":"2008","unstructured":"Barsalou, L.W.: Grounded cognition. Annu. Rev. Psychol. 59, 617\u2013645 (2008)","journal-title":"Annu. Rev. Psychol."},{"key":"4080_CR3","doi-asserted-by":"publisher","first-page":"3871","DOI":"10.1007\/s00371-024-03391-6","volume":"40","author":"SG Ali","year":"2024","unstructured":"Ali, S.G., Zhang, C., Guan, Z., et al.: AI-enhanced digital technologies for myopia management: advancements, challenges, and future prospects. Vis. Comput. 40, 3871\u20133887 (2024)","journal-title":"Vis. Comput."},{"key":"4080_CR4","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: ICML, pp. 8748\u20138763 (2021)"},{"key":"4080_CR5","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1109\/LSP.2023.3346207","volume":"31","author":"S Wang","year":"2024","unstructured":"Wang, S., Yang, D., Zhai, P., Zhang, L.: CPR-CLIP: Multimodal pre-training for composite error recognition in CPR training. IEEE Signal Process. Lett. 31, 211\u2013215 (2024)","journal-title":"IEEE Signal Process. Lett."},{"key":"4080_CR6","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"4080_CR7","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Learning to prompt for vision-language models. Int. J. Comput. Vis. 130, 2337\u20132348 (2022)","journal-title":"Int. J. Comput. Vis."},{"key":"4080_CR8","doi-asserted-by":"crossref","unstructured":"Lin, Z., Yu, S., Kuang, Z., Pathak, D., Ramanan, D.: Multimodality helps unimodality: Cross-modal few-shot learning with multimodal models. In: CVPR, pp. 19325\u201319337 (2023)","DOI":"10.1109\/CVPR52729.2023.01852"},{"key":"4080_CR9","unstructured":"Kimura, A., Ghahramani, Z., Takeuchi, K., Iwata, T., Ueda, N.: Few-shot learning of neural networks from scratch by pseudo example optimization. In: BMVC, p. 105 (2018)"},{"key":"4080_CR10","doi-asserted-by":"crossref","unstructured":"Hariharan, B., Girshick, R.B.: Low-shot visual recognition by shrinking and hallucinating features. In: ICCV, pp. 3037\u20133046 (2017)","DOI":"10.1109\/ICCV.2017.328"},{"key":"4080_CR11","doi-asserted-by":"crossref","unstructured":"Choi, J., Krishnamurthy, J., Kembhavi, A., Farhadi, A.: Structured set matching networks for one-shot part labeling. In: CVPR, pp. 3627\u20133636 (2018)","DOI":"10.1109\/CVPR.2018.00382"},{"key":"4080_CR12","doi-asserted-by":"publisher","DOI":"10.1007\/s00371-024-03592-z","author":"X Chen","year":"2024","unstructured":"Chen, X., Ye, W.: Dual representations network for few-shot learning based on local descriptor importance: integrating global and local features. Vis Comput (2024). https:\/\/doi.org\/10.1007\/s00371-024-03592-z","journal-title":"Vis Comput"},{"key":"4080_CR13","unstructured":"Snell, J., Swersky, K., Zemel, R.: Prototypical networks for few-shot learning. In: NeurIPS, pp. 4080\u20134090 (2017)"},{"key":"4080_CR14","unstructured":"Liu, Y., et al.: Learning to propagate labels: transductive propagation network for few-shot learning. In: ICLR, pp. 1\u201314 (2019)"},{"key":"4080_CR15","doi-asserted-by":"publisher","first-page":"573","DOI":"10.1109\/LSP.2021.3061978","volume":"28","author":"C Xiong","year":"2021","unstructured":"Xiong, C., Li, W., Liu, Y., Wang, M.: Multi-dimensional edge features graph neural network on few-shot image classification. IEEE Signal Process. Lett. 28, 573\u2013577 (2021)","journal-title":"IEEE Signal Process. Lett."},{"key":"4080_CR16","first-page":"1","volume":"18","author":"H Zhao","year":"2024","unstructured":"Zhao, H., Su, Y., Wu, Z., Ding, W.: CSTS: Exploring class-specific and task-shared embedding representation for few-shot learning. IEEE Trans. Neural Netw. Learn. Syst. 18, 1\u201322 (2024)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"4080_CR17","unstructured":"Gao, P., Geng, S., Zhang, R., Ma, T., Fang, R., Zhang, Y., Li, H., Qiao, Y.: Clip-adapter: Better vision-language models with feature adapters. arXiv preprint arXiv:2110.04544(2021)"},{"key":"4080_CR18","doi-asserted-by":"crossref","unstructured":"Pratt, S., Covert, I., Liu, R., et al.: What does a platypus look like? Generating customized prompts for zero-shot image classification. In: ICCV, pp. 15691-15701 (2023)","DOI":"10.1109\/ICCV51070.2023.01438"},{"key":"4080_CR19","doi-asserted-by":"crossref","unstructured":"Ji, A.Y., et al.: Abstract visual reasoning with tangram shapes. In: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, pp. 582\u2013601 (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.38"},{"key":"4080_CR20","unstructured":"Finn, C., Abbeel, P., Levine, S.: Model-agnostic meta-learning for fast adaptation of deep networks. In: ICML, pp. 1126\u20131135 (2017)"},{"key":"4080_CR21","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C. C., Liu, Z.: Conditional prompt learning for vision-language models. In: CVPR, pp. 16795\u201316804 (2022)","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"4080_CR22","doi-asserted-by":"crossref","unstructured":"Wortsman, M., Ilharco, G., Kim, J. W., Li, M., Kornblith, S., Roelofs, R., Schmidt, L.: Robust fine-tuning of zero-shot models. In: CVPR, pp 7959\u20137971 (2022)","DOI":"10.1109\/CVPR52688.2022.00780"},{"issue":"9","key":"4080_CR23","doi-asserted-by":"publisher","first-page":"3341","DOI":"10.1007\/s00371-022-02550-x","volume":"38","author":"Q Liu","year":"2022","unstructured":"Liu, Q., Zhao, J., Cheng, C., Sheng, B., Ma, L.: Pointalcr: adversarial latent GAN and contrastive regularization for point cloud completion. Vis. Comput. 38(9), 3341\u20133349 (2022)","journal-title":"Vis. Comput."},{"key":"4080_CR24","doi-asserted-by":"publisher","first-page":"3507","DOI":"10.1007\/s00371-023-02956-1","volume":"39","author":"S Li","year":"2023","unstructured":"Li, S., Wu, F., Fan, Y., et al.: PLDGAN: portrait line drawing generation with prior knowledge and conditioning target. Vis. Comput. 39, 3507\u20133518 (2023)","journal-title":"Vis. Comput."},{"key":"4080_CR25","doi-asserted-by":"crossref","unstructured":"Lv, S., Guo, D., Xu, J., Tang, D., Duan, N., Gong, M., Shou, L., Jiang, D., Cao, G., Hu, S.: Graph-based reasoning over heterogeneous external knowledge for commonsense question answering. In:AAAI, pp. 8449-8456 (2020)","DOI":"10.1609\/aaai.v34i05.6364"},{"key":"4080_CR26","doi-asserted-by":"crossref","unstructured":"Yu, D., Zhu, C., Yang, Y., Zeng, M.: JAKET: Joint pre-training of knowledge graph and language understanding. In: AAAI, pp. 11630\u201311638 (2022)","DOI":"10.1609\/aaai.v36i10.21417"},{"key":"4080_CR27","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., et al.: ImageNet large scale visual recognition challenge. Int. J. Comput. Vision 115, 211\u2013252 (2015)","journal-title":"Int. J. Comput. Vision"},{"key":"4080_CR28","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"4080_CR29","doi-asserted-by":"crossref","unstructured":"Schoelkopf, B., Herbrich, R., Williamson, R., Smola, A.J.: A generalized representer theorem. In: Proceedings of the 14th Annual Conference on Computational Learning Theory, pp. 416\u2013426 (2001)","DOI":"10.1007\/3-540-44581-1_27"},{"key":"4080_CR30","unstructured":"Soomro, K., Zamir, A.R., Shah, M.: UCF101: a dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402 (2012)"},{"issue":"1","key":"4080_CR31","doi-asserted-by":"publisher","first-page":"59","DOI":"10.1016\/j.cviu.2005.09.012","volume":"106","author":"L Fei-Fei","year":"2007","unstructured":"Fei-Fei, L., Fergus, R., Perona, P.: Learning generative visual models from few training examples: an incremental Bayesian approach tested on 101 object categories. Comput. Vis. Image Underst. 106(1), 59\u201370 (2007)","journal-title":"Comput. Vis. Image Underst."},{"key":"4080_CR32","doi-asserted-by":"crossref","unstructured":"Krause, J., Stark, M., Deng, J., Fei-Fei, L.: 3D object representations for fine-grained categorization. In: Proceedings of the IEEE International Conference on Computer Vision Workshops, pp. 554\u2013561 (2013)","DOI":"10.1109\/ICCVW.2013.77"},{"key":"4080_CR33","doi-asserted-by":"crossref","unstructured":"Nilsback, M.E., Zisserman, A.: Automated flower classification over a large number of classes. In: ICVGIP, pp. 722\u2013729 (2008)","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"4080_CR34","doi-asserted-by":"crossref","unstructured":"Cimpoi, M., Maji, S., Kokkinos, I., Mohamed, S., Vedaldi, A.: Describing textures in the wild. In: CVPR, pp. 3606\u20133613 (2014)","DOI":"10.1109\/CVPR.2014.461"},{"key":"4080_CR35","doi-asserted-by":"crossref","unstructured":"Xiao, J., Hays, J., Ehinger, K.A., Oliva, A., Torralba, A.: Sun database: large-scale scene recognition from abbey to zoo. In: CVPR, pp. 3485\u20133492 (2010)","DOI":"10.1109\/CVPR.2010.5539970"},{"issue":"7","key":"4080_CR36","doi-asserted-by":"publisher","first-page":"2217","DOI":"10.1109\/JSTARS.2019.2918242","volume":"12","author":"P Helber","year":"2019","unstructured":"Helber, P., Bischke, B., Dengel, A., Borth, D.: EuroSAT: a novel dataset and deep learning benchmark for land use and land cover classification. IEEE J. Sel. Top. Appl. Earth Obs. Remote Sens 12(7), 2217\u20132226 (2019)","journal-title":"IEEE J. Sel. Top. Appl. Earth Obs. Remote Sens"},{"key":"4080_CR37","doi-asserted-by":"crossref","unstructured":"Parkhi, O.M., Vedaldi, A., Zisserman, A., Jawahar, C.: Cats and dogs. In: CVPR, pp. 3498\u20133505 (2012)","DOI":"10.1109\/CVPR.2012.6248092"},{"key":"4080_CR38","unstructured":"Maji, S., Rahtu, E., Kannala, J., Blaschko, M., Vedaldi, A.: Fine-grained visual classification of aircraft. arXiv preprint arXiv:1306.5151 (2013)"},{"key":"4080_CR39","unstructured":"Recht, B., Roelofs, R., Schmidt, L., Shankar, V.: Do imagenet classifiers generalize to imagenet? In: ICML, pp. 5389\u20135400 (2019)"},{"key":"4080_CR40","unstructured":"Wang, H., Ge, S., Lipton, Z., Xing, E.P.: Learning robust global representations by penalizing local predictive power. In: NeurIPS, pp. 10506\u201310518 (2019)"},{"key":"4080_CR41","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Zhao, K., Basart, S., Steinhardt, J., Song, D.: Natural adversarial examples. In: CVPR, pp. 15262\u201315271 (2021)","DOI":"10.1109\/CVPR46437.2021.01501"},{"key":"4080_CR42","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., et al.: The many faces of robustness: a critical analysis of out-of-distribution generalization. In: ICCV, pp. 8320\u20138329 (2021)","DOI":"10.1109\/ICCV48922.2021.00823"},{"issue":"11","key":"4080_CR43","first-page":"2579","volume":"9","author":"L Van der Maaten","year":"2008","unstructured":"Van der Maaten, L., Hinton, G.: Visualizing data using t\u2013SNE. J. Mach. Learn. Res. 9(11), 2579\u20132605 (2008)","journal-title":"J. Mach. Learn. Res."}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-04080-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-04080-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-04080-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,24]],"date-time":"2025-09-24T14:00:33Z","timestamp":1758722433000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-04080-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":43,"journal-issue":{"issue":"13","published-print":{"date-parts":[[2025,10]]}},"alternative-id":["4080"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-04080-8","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-5083690\/v1","asserted-by":"object"}]},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,6,30]]},"assertion":[{"value":"17 June 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 June 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}