{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,21]],"date-time":"2026-06-21T11:47:49Z","timestamp":1782042469284,"version":"3.54.5"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2026,6,21]],"date-time":"2026-06-21T00:00:00Z","timestamp":1782000000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,21]],"date-time":"2026-06-21T00:00:00Z","timestamp":1782000000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s00371-026-04514-x","type":"journal-article","created":{"date-parts":[[2026,6,21]],"date-time":"2026-06-21T11:06:33Z","timestamp":1782039993000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Multi-task adaptive CLIP for enhanced person re-identification across domains"],"prefix":"10.1007","volume":"42","author":[{"given":"Yuanyuan","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Simiao","family":"Jia","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rongrong","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dongyan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenjie","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,21]]},"reference":[{"key":"4514_CR1","doi-asserted-by":"publisher","first-page":"23716","DOI":"10.52202\/068431-1723","volume":"35","author":"JB Alayrac","year":"2022","unstructured":"Alayrac, J.B., Donahue, J., Luc, P., Miech, A., Barr, I., Hasson, Y., Lenc, K., Mensch, A., Millican, K., Reynolds, M., et al.: Flamingo: A visual language model for few-shot learning. Adv. Neural. Inf. Process. Syst. 35, 23716\u201323736 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4514_CR2","unstructured":"Bahng, H., Jahanian, A., Sankaranarayanan, S., Isola, P.: Exploring visual prompts for adapting large-scale models. arXiv preprint arXiv:2203.17274 (2022)"},{"key":"4514_CR3","doi-asserted-by":"crossref","unstructured":"Chen, B., Deng, W., Hu, J.: Mixed high-order attention network for person re-identification. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 371\u2013381 (2019)","DOI":"10.1109\/ICCV.2019.00046"},{"key":"4514_CR4","doi-asserted-by":"crossref","unstructured":"Chen, Z., Zhang, Z., Tan, X., Qu, Y., Xie, Y.: Unveiling the power of clip in unsupervised visible-infrared person re-identification. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 3667\u20133675 (2023)","DOI":"10.1145\/3581783.3612050"},{"key":"4514_CR5","doi-asserted-by":"crossref","unstructured":"Deng, W., Zheng, L., Ye, Q., Kang, G., Yang, Y., Jiao, J.: Image-image domain adaptation with preserved self-similarity and domain-dissimilarity for person re-identification. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 994\u20131003 (2018)","DOI":"10.1109\/CVPR.2018.00110"},{"issue":"2","key":"4514_CR6","doi-asserted-by":"publisher","first-page":"581","DOI":"10.1007\/s11263-023-01891-x","volume":"132","author":"P Gao","year":"2024","unstructured":"Gao, P., Geng, S., Zhang, R., Ma, T., Fang, R., Zhang, Y., Li, H., Qiao, Y.: Clip-adapter: Better vision-language models with feature adapters. Int. J. Comput. Vision 132(2), 581\u2013595 (2024)","journal-title":"Int. J. Comput. Vision"},{"key":"4514_CR7","unstructured":"Ge, Y., Chen, D., Li, H.: Mutual mean-teaching: Pseudo label refinery for unsupervised domain adaptation on person re-identification. arXiv preprint arXiv:2001.01526 (2020)"},{"key":"4514_CR8","unstructured":"Gu, X., Lin, T.Y., Kuo, W., Cui, Y.: Open-vocabulary object detection via vision and language knowledge distillation. arXiv preprint arXiv:2104.13921 (2021)"},{"key":"4514_CR9","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proc. of the IEEE conference on computer vision and pattern recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"4514_CR10","doi-asserted-by":"crossref","unstructured":"He, S., Luo, H., Wang, P., Wang, F., Li, H., Jiang, W.: Transreid: Transformer-based object re-identification. In: Proc. of the IEEE\/CVF international conference on computer vision, pp. 15013\u201315022 (2021)","DOI":"10.1109\/ICCV48922.2021.01474"},{"key":"4514_CR11","doi-asserted-by":"crossref","unstructured":"Hou, R., Ma, B., Chang, H., Gu, X., Shan, S., Chen, X.: Interaction-and-aggregation network for person re-identification. In: Proc. of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 9317\u20139326 (2019)","DOI":"10.1109\/CVPR.2019.00954"},{"issue":"9","key":"4514_CR12","first-page":"4894","volume":"44","author":"R Hou","year":"2021","unstructured":"Hou, R., Ma, B., Chang, H., Gu, X., Shan, S., Chen, X.: Feature completion for occluded person re-identification. IEEE Trans. Pattern Anal. Mach. Intell. 44(9), 4894\u20134912 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4514_CR13","doi-asserted-by":"crossref","unstructured":"Hu, M., Zhang, W., Zhou, Q., Wang, R.: Fine-grained text-based person re-identification via interlaced cross-attention and lora fine-tuning: M. hu et al. The Visual Computer pp. 1\u201318 (2025)","DOI":"10.1007\/s00371-025-03931-8"},{"key":"4514_CR14","unstructured":"Jia, C., Yang, Y., Xia, Y., Chen, Y.T., Parekh, Z., Pham, H., Le, Q., Sung, Y.H., Li, Z., Duerig, T.: Scaling up visual and vision-language representation learning with noisy text supervision. In: International conference on machine learning, pp. 4904\u20134916. PMLR (2021)"},{"key":"4514_CR15","doi-asserted-by":"crossref","unstructured":"Jia, M., Tang, L., Chen, B.C., Cardie, C., Belongie, S., Hariharan, B., Lim, S.N.: Visual prompt tuning. In: European conference on computer vision, pp. 709\u2013727. Springer (2022)","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"4514_CR16","doi-asserted-by":"crossref","unstructured":"Khattak, M.U., Rasheed, H., Maaz, M., Khan, S., Khan, F.S.: Maple: Multi-modal prompt learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 19113\u201319122 (2023)","DOI":"10.1109\/CVPR52729.2023.01832"},{"key":"4514_CR17","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems 25 (2012)"},{"key":"4514_CR18","unstructured":"Kumar, A., Ma, T., Liang, P.: Understanding self-training for gradual domain adaptation. In: International conference on machine learning, pp. 5468\u20135479. PMLR (2020)"},{"key":"4514_CR19","doi-asserted-by":"crossref","unstructured":"Li, H., Wu, G., Zheng, W.S.: Combined depth space based architecture search for person re-identification. In: Proc. of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 6729\u20136738 (2021)","DOI":"10.1109\/CVPR46437.2021.00666"},{"key":"4514_CR20","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In: International conference on machine learning, pp. 19730\u201319742. PMLR (2023)"},{"key":"4514_CR21","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International conference on machine learning, pp. 12888\u201312900. PMLR (2022)"},{"key":"4514_CR22","doi-asserted-by":"crossref","unstructured":"Li, S., Sun, L., Li, Q.: Clip-reid: exploiting vision-language model for image re-identification without concrete text labels. In: Proc. of the AAAI conference on artificial intelligence, vol.\u00a037, pp. 1405\u20131413 (2023)","DOI":"10.1609\/aaai.v37i1.25225"},{"key":"4514_CR23","doi-asserted-by":"crossref","unstructured":"Luo, H., Gu, Y., Liao, X., Lai, S., Jiang, W.: Bag of tricks and a strong baseline for deep person re-identification. In: Proc. of the IEEE\/CVF conference on computer vision and pattern recognition workshops, pp. 0\u201310 (2019)","DOI":"10.1109\/CVPRW.2019.00190"},{"key":"4514_CR24","doi-asserted-by":"crossref","unstructured":"Qian, X., Wang, W., Zhang, L., Zhu, F., Fu, Y., Xiang, T., Jiang, Y.G., Xue, X.: Long-term cloth-changing person re-identification. In: Proc. of the Asian conference on computer vision (2020)","DOI":"10.1007\/978-3-030-69535-4_5"},{"key":"4514_CR25","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International conference on machine learning, pp. 8748\u20138763. PmLR (2021)"},{"key":"4514_CR26","unstructured":"Ramesh, A., Pavlov, M., Goh, G., Gray, S., Voss, C., Radford, A., Chen, M., Sutskever, I.: Zero-shot text-to-image generation. In: International conference on machine learning, pp. 8821\u20138831. Pmlr (2021)"},{"key":"4514_CR27","doi-asserted-by":"crossref","unstructured":"Ristani, E., Solera, F., Zou, R., Cucchiara, R., Tomasi, C.: Performance measures and a data set for multi-target, multi-camera tracking. In: European Conference on Computer Vision, pp. 17\u201335. Springer (2016)","DOI":"10.1007\/978-3-319-48881-3_2"},{"key":"4514_CR28","doi-asserted-by":"crossref","unstructured":"Sun, Y., Zheng, L., Yang, Y., Tian, Q., Wang, S.: Beyond part models: Person retrieval with refined part pooling (and a strong convolutional baseline). In: Proc. of the European conference on computer vision (ECCV), pp. 480\u2013496 (2018)","DOI":"10.1007\/978-3-030-01225-0_30"},{"key":"4514_CR29","unstructured":"Tan, L., Dai, P., Chen, J., Cao, L., Wu, Y., Ji, R.: Partformer: Awakening latent diverse representation from vision transformer for object re-identification. arXiv preprint arXiv:2408.16684 (2024)"},{"key":"4514_CR30","unstructured":"Touvron, H., Cord, M., Douze, M., Massa, F., Sablayrolles, A., J\u00e9gou, H.: Training data-efficient image transformers & distillation through attention. In: International conference on machine learning, pp. 10347\u201310357. PMLR (2021)"},{"key":"4514_CR31","first-page":"200","volume":"34","author":"M Tsimpoukelli","year":"2021","unstructured":"Tsimpoukelli, M., Menick, J.L., Cabi, S., Eslami, S., Vinyals, O., Hill, F.: Multimodal few-shot learning with frozen language models. Adv. Neural. Inf. Process. Syst. 34, 200\u2013212 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4514_CR32","doi-asserted-by":"crossref","unstructured":"Wang, G., Yuan, Y., Chen, X., Li, J., Zhou, X.: Learning discriminative features with multiple granularities for person re-identification. In: Proc. of the 26th ACM international conference on Multimedia, pp. 274\u2013282 (2018)","DOI":"10.1145\/3240508.3240552"},{"key":"4514_CR33","doi-asserted-by":"crossref","unstructured":"Wang, J., Zhu, X., Gong, S., Li, W.: Transferable joint attribute-identity deep learning for unsupervised person re-identification. In: Proc. of the IEEE conference on computer vision and pattern recognition, pp. 2275\u20132284 (2018)","DOI":"10.1109\/CVPR.2018.00242"},{"issue":"6","key":"4514_CR34","doi-asserted-by":"publisher","first-page":"509","DOI":"10.1016\/j.vrih.2023.06.003","volume":"5","author":"M Wang","year":"2023","unstructured":"Wang, M., et al.: Adequate alignment and interaction for cross-modal retrieval. Virtual Real. Intell. Hardware 5(6), 509\u2013522 (2023)","journal-title":"Virtual Real. Intell. Hardware"},{"key":"4514_CR35","first-page":"11960","volume":"34","author":"Y Wang","year":"2021","unstructured":"Wang, Y., Huang, R., Song, S., Huang, Z., Huang, G.: Not all images are worth 16x16 words: Dynamic transformers for efficient image recognition. Adv. Neural. Inf. Process. Syst. 34, 11960\u201311973 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4514_CR36","doi-asserted-by":"crossref","unstructured":"Wei, L., Zhang, S., Gao, W., Tian, Q.: Person transfer gan to bridge domain gap for person re-identification. In: Proc. of the IEEE conference on computer vision and pattern recognition, pp. 79\u201388 (2018)","DOI":"10.1109\/CVPR.2018.00016"},{"key":"4514_CR37","doi-asserted-by":"crossref","unstructured":"Woo, S., Park, J., Lee, J.Y., Kweon, I.S.: Cbam: Convolutional block attention module. In: Proceedings of the European conference on computer vision (ECCV), pp. 3\u201319 (2018)","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"4514_CR38","doi-asserted-by":"crossref","unstructured":"Yao, H., Zhang, R., Xu, C.: Visual-language prompt tuning with knowledge-guided context optimization. In: Proc. of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 6757\u20136767 (2023)","DOI":"10.1109\/CVPR52729.2023.00653"},{"issue":"6","key":"4514_CR39","doi-asserted-by":"publisher","first-page":"2872","DOI":"10.1109\/TPAMI.2021.3054775","volume":"44","author":"M Ye","year":"2021","unstructured":"Ye, M., Shen, J., Lin, G., Xiang, T., Shao, L., Hoi, S.C.: Deep learning for person re-identification: A survey and outlook. IEEE Trans. Pattern Anal. Mach. Intell. 44(6), 2872\u20132893 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4514_CR40","doi-asserted-by":"crossref","unstructured":"Yu, C., Liu, X., Wang, Y., Zhang, P., Lu, H.: Tf-clip: Learning text-free clip for video-based person re-identification. In: Proc. of the AAAI conference on artificial intelligence, vol.\u00a038, pp. 6764\u20136772 (2024)","DOI":"10.1609\/aaai.v38i7.28500"},{"key":"4514_CR41","unstructured":"Yu, J., Wang, Z., Vasudevan, V., Yeung, L., Seyedhosseini, M., Wu, Y.: Coca: Contrastive captioners are image-text foundation models. arXiv preprint arXiv:2205.01917 (2022)"},{"key":"4514_CR42","unstructured":"Yuan, L., Chen, D., Chen, Y.L., Codella, N., Dai, X., Gao, J., Hu, H., Huang, X., Li, B., Li, C., et\u00a0al.: Florence: A new foundation model for computer vision. arXiv preprint arXiv:2111.11432 (2021)"},{"key":"4514_CR43","doi-asserted-by":"crossref","unstructured":"Zeng, W., Ge, H., Liu, Y., Li, B.: Ought to be salient and hidden: soft biological semantic-guided explicit\u2013implicit learning for cloth-changing person re-identification: W. zeng et al. The Visual Computer 41(14), 12189\u201312204 (2025)","DOI":"10.1007\/s00371-025-04151-w"},{"key":"4514_CR44","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Lan, C., Zeng, W., Jin, X., Chen, Z.: Relation-aware global attention for person re-identification. In: Proc. of the ieee\/cvf conference on computer vision and pattern recognition, pp. 3186\u20133195 (2020)","DOI":"10.1109\/CVPR42600.2020.00325"},{"key":"4514_CR45","doi-asserted-by":"crossref","unstructured":"Zheng, L., Shen, L., Tian, L., Wang, S., Wang, J., Tian, Q.: Scalable person re-identification: A benchmark. In: Proc. of the IEEE international conference on computer vision, pp. 1116\u20131124 (2015)","DOI":"10.1109\/ICCV.2015.133"},{"key":"4514_CR46","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Conditional prompt learning for vision-language models. In: Proc. of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 16816\u201316825 (2022)","DOI":"10.1109\/CVPR52688.2022.01631"},{"issue":"9","key":"4514_CR47","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Learning to prompt for vision-language models. Int. J. Comput. Vision 130(9), 2337\u20132348 (2022)","journal-title":"Int. J. Comput. Vision"},{"key":"4514_CR48","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, Y., Cavallaro, A., Xiang, T.: Omni-scale feature learning for person re-identification. In: Proc. of the IEEE\/CVF international conference on computer vision, pp. 3702\u20133712 (2019)","DOI":"10.1109\/ICCV.2019.00380"},{"key":"4514_CR49","doi-asserted-by":"crossref","unstructured":"Zhu, B., Niu, Y., Han, Y., Wu, Y., Zhang, H.: Prompt-aligned gradient for prompt tuning. In: Proc. of the IEEE\/CVF international conference on computer vision, pp. 15659\u201315669 (2023)","DOI":"10.1109\/ICCV51070.2023.01435"},{"key":"4514_CR50","doi-asserted-by":"crossref","unstructured":"Zhu, H., Ke, W., Li, D., Liu, J., Tian, L., Shan, Y.: Dual cross-attention learning for fine-grained visual categorization and object re-identification. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 4692\u20134702 (2022)","DOI":"10.1109\/CVPR52688.2022.00465"},{"key":"4514_CR51","doi-asserted-by":"crossref","unstructured":"Zhu, K., Guo, H., Zhang, S., Wang, Y., Liu, J., Wang, J., Tang, M.: Aaformer: Auto-aligned transformer for person re-identification. IEEE Transac. Neural Netw. Learn. Syst. (2023)","DOI":"10.1109\/TNNLS.2023.3301856"},{"key":"4514_CR52","doi-asserted-by":"publisher","first-page":"45666","DOI":"10.52202\/079017-1452","volume":"37","author":"J Zuo","year":"2024","unstructured":"Zuo, J., Hong, J., Zhang, F., Yu, C., Zhou, H., Gao, C., Sang, N., Wang, J.: Plip: Language-image pre-training for person representation learning. Adv. Neural. Inf. Process. Syst. 37, 45666\u201345702 (2024)","journal-title":"Adv. Neural. Inf. Process. Syst."}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04514-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-026-04514-x","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04514-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,21]],"date-time":"2026-06-21T11:06:48Z","timestamp":1782040008000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-026-04514-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,21]]},"references-count":52,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["4514"],"URL":"https:\/\/doi.org\/10.1007\/s00371-026-04514-x","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6,21]]},"assertion":[{"value":"22 January 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 April 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 June 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"352"}}