{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T18:20:28Z","timestamp":1783189228267,"version":"3.54.6"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T00:00:00Z","timestamp":1743120000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T00:00:00Z","timestamp":1743120000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U1931202"],"award-info":[{"award-number":["U1931202"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62076033"],"award-info":[{"award-number":["62076033"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,5]]},"DOI":"10.1007\/s10489-025-06487-1","type":"journal-article","created":{"date-parts":[[2025,3,30]],"date-time":"2025-03-30T23:09:48Z","timestamp":1743376188000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Automated text annotation: a new paradigm for generalizable text-to-image person retrieval"],"prefix":"10.1007","volume":"55","author":[{"given":"Delong","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6506-7298","authenticated-orcid":false,"given":"Zhicheng","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fei","family":"Su","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,3,28]]},"reference":[{"key":"6487_CR1","doi-asserted-by":"crossref","unstructured":"Reed S, Akata Z, Lee H et\u00a0al (2016) Learning deep representations of fine-grained visual descriptions. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 49\u201358","DOI":"10.1109\/CVPR.2016.13"},{"key":"6487_CR2","unstructured":"Radford A, Kim JW, Hallacy C et\u00a0al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning. PMLR, pp 8748\u20138763"},{"key":"6487_CR3","unstructured":"Jia C, Yang Y, Xia Y et\u00a0al (2021) Scaling up visual and vision-language representation learning with noisy text supervision. In: International conference on machine learning. PMLR, pp 4904\u20134916"},{"key":"6487_CR4","doi-asserted-by":"crossref","unstructured":"Li S, Xiao T, Li H et\u00a0al (2017) Person search with natural language description. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1970\u20131979","DOI":"10.1109\/CVPR.2017.551"},{"key":"6487_CR5","doi-asserted-by":"crossref","unstructured":"Farooq A, Awais M, Kittler J et\u00a0al (2022) Axm-net: implicit cross-modal feature alignment for person re-identification. In: Proceedings of the AAAI conference on artificial intelligence, pp 4477\u20134485","DOI":"10.1609\/aaai.v36i4.20370"},{"issue":"2","key":"6487_CR6","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3383184","volume":"16","author":"Z Zheng","year":"2020","unstructured":"Zheng Z, Zheng L, Garrett M et al (2020) Dual-path convolutional image-text embeddings with instance loss. ACM Trans Multimed Comput Commun Appl (TOMM) 16(2):1\u201323","journal-title":"ACM Trans Multimed Comput Commun Appl (TOMM)"},{"key":"6487_CR7","doi-asserted-by":"crossref","unstructured":"He S, Luo H, Wang P et\u00a0al (2021) Transreid: transformer-based object re-identification. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 15013\u201315022","DOI":"10.1109\/ICCV48922.2021.01474"},{"key":"6487_CR8","unstructured":"Ding Z, Ding C, Shao Z et\u00a0al (2021) Semantically self-aligned network for text-to-image part-aware person re-identification. arXiv:2107.12666"},{"key":"6487_CR9","doi-asserted-by":"crossref","unstructured":"Zhu A, Wang Z, Li Y et\u00a0al (2021) Dssl: deep surroundings-person separation learning for text-based person retrieval. In: Proceedings of the 29th ACM international conference on multimedia, pp 209\u2013217","DOI":"10.1145\/3474085.3475369"},{"key":"6487_CR10","doi-asserted-by":"crossref","unstructured":"Sarafianos N, Xu X, Kakadiaris IA (2019) Adversarial representation learning for text-to-image matching. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 5814\u20135824","DOI":"10.1109\/ICCV.2019.00591"},{"key":"6487_CR11","doi-asserted-by":"publisher","first-page":"171","DOI":"10.1016\/j.neucom.2022.04.081","volume":"494","author":"Y Chen","year":"2022","unstructured":"Chen Y, Zhang G, Lu Y et al (2022) Tipcb: a simple but effective part-based convolutional baseline for text-based person search. Neurocomputing 494:171\u2013181","journal-title":"Neurocomputing"},{"issue":"23","key":"6487_CR12","doi-asserted-by":"publisher","first-page":"29183","DOI":"10.1007\/s10489-023-05039-9","volume":"53","author":"Z Wang","year":"2023","unstructured":"Wang Z, Ye X, Shang X et al (2023) Person re-identification method with mahalanobis trm triplet on multi-branch network. Appl Intell 53(23):29183\u201329204","journal-title":"Appl Intell"},{"issue":"3","key":"6487_CR13","doi-asserted-by":"publisher","first-page":"2656","DOI":"10.1007\/s10489-022-03570-9","volume":"53","author":"J Liu","year":"2023","unstructured":"Liu J, Lin M, Zhao M et al (2023) Person re-identification via semi-supervised adaptive graph embedding. Appl Intell 53(3):2656\u20132672","journal-title":"Appl Intell"},{"key":"6487_CR14","doi-asserted-by":"crossref","unstructured":"Shu X, Wen W, Wu H et\u00a0al (2022) See finer, see more: implicit modality alignment for text-based person retrieval. In: European conference on computer vision. Springer, pp 624\u2013641","DOI":"10.1007\/978-3-031-25072-9_42"},{"key":"6487_CR15","doi-asserted-by":"crossref","unstructured":"Zhang Y, Lu H (2018) Deep cross-modal projection learning for image-text matching. In: Proceedings of the European conference on computer vision (ECCV), pp 686\u2013701","DOI":"10.1007\/978-3-030-01246-5_42"},{"key":"6487_CR16","doi-asserted-by":"crossref","unstructured":"Yan S, Dong N, Zhang L et\u00a0al (2023) Clip-driven fine-grained text-image person re-identification. IEEE Trans Image Process","DOI":"10.1109\/TIP.2023.3327924"},{"key":"6487_CR17","doi-asserted-by":"crossref","unstructured":"Wang Z, Zhu A, Xue J et\u00a0al (2022) Look before you leap: improving text-based person retrieval by learning a consistent cross-modal common manifold. In: Proceedings of the 30th ACM international conference on multimedia, pp 1984\u20131992","DOI":"10.1145\/3503161.3548166"},{"key":"6487_CR18","doi-asserted-by":"publisher","first-page":"6079","DOI":"10.1109\/TMM.2022.3204444","volume":"25","author":"X Wang","year":"2022","unstructured":"Wang X, Zhu L, Zheng Z et al (2022) Align and tell: boosting text-video retrieval with local alignment and fine-grained supervision. IEEE Trans Multimed 25:6079\u20136089","journal-title":"IEEE Trans Multimed"},{"key":"6487_CR19","doi-asserted-by":"crossref","unstructured":"Wu Y, Yan Z, Han X et\u00a0al (2021) Lapscore: language-guided person search via color reasoning. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 1624\u20131633","DOI":"10.1109\/ICCV48922.2021.00165"},{"key":"6487_CR20","doi-asserted-by":"crossref","unstructured":"Jiang D, Ye M (2023) Cross-modal implicit relation reasoning and aligning for text-to-image person retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 2787\u20132797","DOI":"10.1109\/CVPR52729.2023.00273"},{"key":"6487_CR21","doi-asserted-by":"publisher","unstructured":"Liu D, Li H, Zhao Z et al (2025) Text-guided image restoration and semantic enhancement for text-to-image person retrieval. Neural Netw 184:107028. https:\/\/doi.org\/10.1016\/j.neunet.2024.107028. https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0893608024009572","DOI":"10.1016\/j.neunet.2024.107028"},{"key":"6487_CR22","doi-asserted-by":"crossref","unstructured":"Ma Z, Chen H, Zeng W et\u00a0al (2025) Multi-modal reference learning for fine-grained text-to-image retrieval. IEEE Trans Multimed","DOI":"10.1109\/TMM.2025.3543066"},{"issue":"3\u20134","key":"6487_CR23","doi-asserted-by":"publisher","first-page":"163","DOI":"10.1561\/0600000105","volume":"14","author":"Z Gan","year":"2022","unstructured":"Gan Z, Li L, Li C et al (2022) Vision-language pre-training: basics, recent advances, and future trends. Found Trends\u00ae Comput Graph Vis 14(3\u20134):163\u2013352","journal-title":"Found Trends\u00ae Comput Graph Vis"},{"issue":"23","key":"6487_CR24","doi-asserted-by":"publisher","first-page":"12230","DOI":"10.1007\/s10489-024-05823-1","volume":"54","author":"T Yao","year":"2024","unstructured":"Yao T, Peng S, Wang L et al (2024) Cross-modality interaction reasoning for enhancing vision-language pre-training in image-text retrieval. Appl Intell 54(23):12230\u201312245","journal-title":"Appl Intell"},{"key":"6487_CR25","doi-asserted-by":"crossref","unstructured":"Huang Z, Zeng Z, Huang Y et\u00a0al (2021) Seeing out of the box: end-to-end pre-training for vision-language representation learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12976\u201312985","DOI":"10.1109\/CVPR46437.2021.01278"},{"key":"6487_CR26","doi-asserted-by":"crossref","unstructured":"Li X, Yin X, Li C et\u00a0al (2020) Oscar: object-semantics aligned pre-training for vision-language tasks. In: Computer vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXX 16. Springer, pp 121\u2013137","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"6487_CR27","doi-asserted-by":"publisher","unstructured":"Suhr A, Zhou S, Zhang A et\u00a0al (2019) A corpus for reasoning about natural language grounded in photographs. In: Korhonen A, Traum DR, M\u00e0rquez L (eds) Proceedings of the 57th conference of the association for computational linguistics, ACL 2019, Florence, Italy, July 28- August 2, 2019, Volume 1: Long Papers. Association for Computational Linguistics, pp 6418\u20136428. https:\/\/doi.org\/10.18653\/V1\/P19-1644","DOI":"10.18653\/V1\/P19-1644"},{"key":"6487_CR28","doi-asserted-by":"crossref","unstructured":"Antol S, Agrawal A, Lu J et\u00a0al (2015) Vqa: visual question answering. In: Proceedings of the IEEE international conference on computer vision, pp 2425\u20132433","DOI":"10.1109\/ICCV.2015.279"},{"key":"6487_CR29","unstructured":"Ramesh A, Pavlov M, Goh G et\u00a0al (2021) Zero-shot text-to-image generation. In: International conference on machine learning. PMLR, pp 8821\u20138831"},{"key":"6487_CR30","doi-asserted-by":"crossref","unstructured":"He K, Fan H, Wu Y et\u00a0al (2020) Momentum contrast for unsupervised visual representation learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9729\u20139738","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"6487_CR31","unstructured":"Chen T, Kornblith S, Norouzi M et\u00a0al (2020) A simple framework for contrastive learning of visual representations. In: International conference on machine learning. PMLR, pp 1597\u20131607"},{"key":"6487_CR32","unstructured":"Li J, Selvaraju R, Gotmare A et al (2021) Align before fuse: vision and language representation learning with momentum distillation. Advances in neural information processing systems, vol 34, pp 9694\u20139705"},{"key":"6487_CR33","unstructured":"Li J, Li D, Xiong C et\u00a0al (2022) Blip: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning. PMLR, pp 12888\u201312900"},{"key":"6487_CR34","doi-asserted-by":"publisher","first-page":"103075","DOI":"10.1016\/j.aei.2024.103075","volume":"64","author":"R Cai","year":"2025","unstructured":"Cai R, Guo Z, Chen X et al (2025) Automatic identification of integrated construction elements using open-set object detection based on image and text modality fusion. Adv Eng Inform 64:103075","journal-title":"Adv Eng Inform"},{"key":"6487_CR35","doi-asserted-by":"publisher","first-page":"1389","DOI":"10.1109\/TIP.2024.3364495","volume":"33","author":"B Su","year":"2024","unstructured":"Su B, Zhang H, Li J et al (2024) Toward generalized few-shot open-set object detection. IEEE Trans Image Process 33:1389\u20131402","journal-title":"IEEE Trans Image Process"},{"key":"6487_CR36","doi-asserted-by":"crossref","unstructured":"Zareian A, Rosa KD, Hu DH et\u00a0al (2021) Open-vocabulary object detection using captions. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 14393\u201314402","DOI":"10.1109\/CVPR46437.2021.01416"},{"key":"6487_CR37","doi-asserted-by":"crossref","unstructured":"Carion N, Massa F, Synnaeve G et\u00a0al (2020) End-to-end object detection with transformers. In: European conference on computer vision. Springer, pp 213\u2013229","DOI":"10.1007\/978-3-030-58452-8_13"},{"issue":"2","key":"6487_CR38","doi-asserted-by":"publisher","first-page":"581","DOI":"10.1007\/s11263-023-01891-x","volume":"132","author":"P Gao","year":"2024","unstructured":"Gao P, Geng S, Zhang R et al (2024) Clip-adapter: better vision-language models with feature adapters. Int J Comput Vis 132(2):581\u2013595","journal-title":"Int J Comput Vis"},{"key":"6487_CR39","unstructured":"Yao L, Han J, Wen Y et al (2022) Detclip: dictionary-enriched visual-concept paralleled pre-training for open-world detection. Advances in neural information processing systems, vol 35, pp 9125\u20139138"},{"key":"6487_CR40","doi-asserted-by":"crossref","unstructured":"Liu S, Zeng Z, Ren T et\u00a0al (2024) Grounding dino: marrying dino with grounded pre-training for open-set object detection. European conference on computer vision","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"6487_CR41","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A et\u00a0al (2021) An image is worth 16x16 words: transformers for image recognition at scale. In: Proceedings of the International Conference on Learning Representations (ICLR)"},{"key":"6487_CR42","unstructured":"Devlin J, Chang MW, Lee K et\u00a0al (2019) Bert: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: Human language technologies. Association for Computational Linguistics, pp 4171\u20134186"},{"key":"6487_CR43","doi-asserted-by":"crossref","unstructured":"Anderson P, He X, Buehler C et\u00a0al (2018) Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6077\u20136086","DOI":"10.1109\/CVPR.2018.00636"},{"key":"6487_CR44","unstructured":"Holtzman A, Buys J, Du L et\u00a0al (2020) The curious case of neural text degeneration. In: Proceedings of the International Conference on Learning Representations (ICLR)"},{"key":"6487_CR45","doi-asserted-by":"crossref","unstructured":"Wei L, Zhang S, Gao W et\u00a0al (2018) Person transfer gan to bridge domain gap for person re-identification. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 79\u201388","DOI":"10.1109\/CVPR.2018.00016"},{"key":"6487_CR46","doi-asserted-by":"crossref","unstructured":"Jing Y, Si C, Wang J et\u00a0al (2020) Pose-guided multi-granularity attention network for text-based person search. In: Proceedings of the AAAI conference on artificial intelligence, pp 11189\u201311196","DOI":"10.1609\/aaai.v34i07.6777"},{"key":"6487_CR47","doi-asserted-by":"crossref","unstructured":"Wang Z, Fang Z, Wang J et\u00a0al (2020) Vitaa: Visual-textual attributes alignment in person search by natural language. In: Computer vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XII 16. Springer, pp 402\u2013420","DOI":"10.1007\/978-3-030-58610-2_24"},{"key":"6487_CR48","doi-asserted-by":"publisher","unstructured":"Zhu A, Wang Z, Xue J et\u00a0al (2024) Improving text-based person retrieval by excavating all-round information beyond color. IEEE Trans Neural Netw Learn Syst 1\u201315. https:\/\/doi.org\/10.1109\/TNNLS.2024.3368217","DOI":"10.1109\/TNNLS.2024.3368217"},{"key":"6487_CR49","doi-asserted-by":"publisher","first-page":"1990","DOI":"10.1109\/TIP.2024.3372832","volume":"33","author":"K Niu","year":"2024","unstructured":"Niu K, Huang L, Long Y et al (2024) Comprehensive attribute prediction learning for person search by language. IEEE Trans Image Process 33:1990\u20132003. https:\/\/doi.org\/10.1109\/TIP.2024.3372832","journal-title":"IEEE Trans Image Process"},{"key":"6487_CR50","doi-asserted-by":"publisher","first-page":"4281","DOI":"10.1109\/TMM.2023.3321504","volume":"26","author":"L Bao","year":"2024","unstructured":"Bao L, Wei L, Zhou W et al (2024) Multi-granularity matching transformer for text-based person search. IEEE Trans Multimed 26:4281\u20134293. https:\/\/doi.org\/10.1109\/TMM.2023.3321504","journal-title":"IEEE Trans Multimed"},{"issue":"9","key":"6487_CR51","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou K, Yang J, Loy CC et al (2022) Learning to prompt for vision-language models. Int J Comput Vis 130(9):2337\u20132348","journal-title":"Int J Comput Vis"},{"key":"6487_CR52","doi-asserted-by":"crossref","unstructured":"Zhou K, Yang J, Loy CC et\u00a0al (2022) Conditional prompt learning for vision-language models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 16816\u201316825","DOI":"10.1109\/CVPR52688.2022.01631"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06487-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-025-06487-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06487-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,19]],"date-time":"2025-09-19T19:32:26Z","timestamp":1758310346000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-025-06487-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,28]]},"references-count":52,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2025,5]]}},"alternative-id":["6487"],"URL":"https:\/\/doi.org\/10.1007\/s10489-025-06487-1","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,3,28]]},"assertion":[{"value":"19 March 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 March 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"593"}}