{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T19:16:02Z","timestamp":1783106162817,"version":"3.54.6"},"reference-count":37,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62206006"],"award-info":[{"award-number":["62206006"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100009091","name":"Justus Liebig Universit\u00e4t Gie\u00dfen","doi-asserted-by":"publisher","award":["93K172025K21"],"award-info":[{"award-number":["93K172025K21"]}],"id":[{"id":"10.13039\/100009091","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100006248","name":"Anhui University of Technology","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100006248","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003995","name":"Natural Science Foundation of Anhui Province","doi-asserted-by":"publisher","award":["2024AH051783"],"award-info":[{"award-number":["2024AH051783"]}],"id":[{"id":"10.13039\/501100003995","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Image and Vision Computing"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.imavis.2026.105999","type":"journal-article","created":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T06:53:11Z","timestamp":1776840791000},"page":"105999","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Unsupervised cross-modal person search via text style normalization"],"prefix":"10.1016","volume":"171","author":[{"given":"Feng","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jielong","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fuzhong","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jun","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bin","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6562-6018","authenticated-orcid":false,"given":"Wei","family":"Huo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.imavis.2026.105999_b1","doi-asserted-by":"crossref","unstructured":"D. Jiang, M. Ye, Cross-modal implicit relation reasoning and aligning for text-to-image person retrieval, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 2787\u20132797.","DOI":"10.1109\/CVPR52729.2023.00273"},{"issue":"6","key":"10.1016\/j.imavis.2026.105999_b2","doi-asserted-by":"crossref","first-page":"5118","DOI":"10.1109\/TCSVT.2023.3340102","article-title":"Multiple information embedded hashing for large-scale cross-modal retrieval","volume":"34","author":"Wang","year":"2023","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.imavis.2026.105999_b3","article-title":"MBDBFormer: a multimodal bridge dual branch transformer for person re-identification","author":"Deng","year":"2025","journal-title":"Digit. Signal Process."},{"key":"10.1016\/j.imavis.2026.105999_b4","doi-asserted-by":"crossref","first-page":"6609","DOI":"10.1109\/TMM.2024.3355644","article-title":"Cross-modal adaptive dual association for text-to-image person retrieval","volume":"26","author":"Lin","year":"2024","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.imavis.2026.105999_b5","unstructured":"T. Gong, J. Wang, L. Zhang, Enhancing cross-modal completion and alignment for unsupervised incomplete text-to-image person retrieval, in: Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence, 2024, pp. 794\u2013802."},{"key":"10.1016\/j.imavis.2026.105999_b6","unstructured":"Z. Li, J. Li, Y. Shi, H. Ling, J. Chen, R. Wang, S. Huang, Cross-modal generation and alignment via attribute-guided prompt for unsupervised text-based person retrieval, in: Proceedings of the International Joint Conference on Artificial Intelligence. International Joint Conferences on Artificial Intelligence Organization, 2024, pp. 1047\u20131055."},{"key":"10.1016\/j.imavis.2026.105999_b7","doi-asserted-by":"crossref","first-page":"151","DOI":"10.1016\/j.patcog.2019.06.006","article-title":"Improving person re-identification by attribute and identity learning","volume":"95","author":"Lin","year":"2019","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.imavis.2026.105999_b8","first-page":"6979","article-title":"Multi-prompts learning with cross-modal alignment for attribute-based person re-identification","volume":"vol. 38","author":"Zhai","year":"2024"},{"key":"10.1016\/j.imavis.2026.105999_b9","series-title":"Qwen2-vl: Enhancing vision-language model\u2019s perception of the world at any resolution","author":"Wang","year":"2024"},{"key":"10.1016\/j.imavis.2026.105999_b10","doi-asserted-by":"crossref","unstructured":"N. Rotstein, D. Bensaid, S. Brody, R. Ganz, R. Kimmel, Fusecap: Leveraging large language models for enriched fused image captions, in: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, 2024, pp. 5689\u20135700.","DOI":"10.1109\/WACV57701.2024.00559"},{"key":"10.1016\/j.imavis.2026.105999_b11","series-title":"International Conference on Machine Learning","first-page":"12888","article-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"},{"key":"10.1016\/j.imavis.2026.105999_b12","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109369","article-title":"Unsupervised person re-identification via multi-domain joint learning","volume":"138","author":"Chen","year":"2023","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.imavis.2026.105999_b13","doi-asserted-by":"crossref","unstructured":"S. Yang, Y. Zhou, Z. Zheng, Y. Wang, L. Zhu, Y. Wu, Towards unified text-based person retrieval: A large-scale multi-attribute and language search benchmark, in: Proceedings of the 31st ACM International Conference on Multimedia, 2023, pp. 4492\u20134501.","DOI":"10.1145\/3581783.3611709"},{"key":"10.1016\/j.imavis.2026.105999_b14","doi-asserted-by":"crossref","unstructured":"W. Tan, C. Ding, J. Jiang, F. Wang, Y. Zhan, D. Tao, Harnessing the power of mllms for transferable text-to-image person reid, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 17127\u201317137.","DOI":"10.1109\/CVPR52733.2024.01621"},{"key":"10.1016\/j.imavis.2026.105999_b15","doi-asserted-by":"crossref","unstructured":"J. Jiang, C. Ding, W. Tan, J. Wang, J. Tao, X. Xu, Modeling Thousands of Human Annotators for Generalizable Text-to-Image Person Re-identification, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 9220\u20139230.","DOI":"10.1109\/CVPR52734.2025.00861"},{"key":"10.1016\/j.imavis.2026.105999_b16","doi-asserted-by":"crossref","unstructured":"F. Chen, J. He, Y. Liu, H. Liu, Z. Chen, Y. Wang, Unsupervised Cross-Modal Person Search via Progressive Diverse Text Generation, in: Proceedings of the 33rd ACM International Conference on Multimedia, 2025, pp. 6047\u20136056.","DOI":"10.1145\/3746027.3755171"},{"key":"10.1016\/j.imavis.2026.105999_b17","doi-asserted-by":"crossref","unstructured":"Y. Qin, C. Chen, Z. Fu, D. Peng, X. Peng, P. Hu, Human-centered interactive learning via mllms for text-to-image person re-identification, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 14390\u201314399.","DOI":"10.1109\/CVPR52734.2025.01342"},{"key":"10.1016\/j.imavis.2026.105999_b18","doi-asserted-by":"crossref","unstructured":"Z. Dai, G. Wang, W. Yuan, S. Zhu, P. Tan, Cluster contrast for unsupervised person re-identification, in: Proceedings of the Asian Conference on Computer Vision, 2022, pp. 1142\u20131160.","DOI":"10.1007\/978-3-031-26351-4_20"},{"key":"10.1016\/j.imavis.2026.105999_b19","series-title":"European Conference on Computer Vision","first-page":"20","article-title":"An attention-driven two-stage clustering method for unsupervised person re-identification","author":"Ji","year":"2020"},{"key":"10.1016\/j.imavis.2026.105999_b20","doi-asserted-by":"crossref","unstructured":"Y. Tu, B. Zhang, Y. Li, L. Liu, J. Li, J. Zhang, Y. Wang, C. Wang, C.R. Zhao, Learning with noisy labels via self-supervised adversarial noisy masking, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 16186\u201316195.","DOI":"10.1109\/CVPR52729.2023.01553"},{"key":"10.1016\/j.imavis.2026.105999_b21","series-title":"Chinese Conference on Pattern Recognition and Computer Vision","first-page":"468","article-title":"Multimodal feature hierarchical fusion for text-image person re-identification","author":"Li","year":"2024"},{"key":"10.1016\/j.imavis.2026.105999_b22","doi-asserted-by":"crossref","unstructured":"Y. Qin, Y. Chen, D. Peng, X. Peng, J.T. Zhou, P. Hu, Noisy-correspondence learning for text-to-image person re-identification, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 27197\u201327206.","DOI":"10.1109\/CVPR52733.2024.02568"},{"key":"10.1016\/j.imavis.2026.105999_b23","doi-asserted-by":"crossref","unstructured":"S. Yan, J. Liu, N. Dong, L. Zhang, J. Tang, Cross-modal collaborative representation learning for text-to-image person retrieval, in: Proceedings of the Thirty-Fourth International Joint Conference on Artificial Intelligence, IJCAI-25, 2025.","DOI":"10.24963\/ijcai.2025\/240"},{"key":"10.1016\/j.imavis.2026.105999_b24","first-page":"568","article-title":"Graph-based cross-domain knowledge distillation for cross-dataset text-to-image person retrieval","volume":"vol. 39","author":"Luo","year":"2025"},{"key":"10.1016\/j.imavis.2026.105999_b25","first-page":"226","article-title":"A density-based algorithm for discovering clusters in large spatial databases with noise","volume":"vol. 96","author":"Ester","year":"1996"},{"key":"10.1016\/j.imavis.2026.105999_b26","doi-asserted-by":"crossref","unstructured":"S. Li, T. Xiao, H. Li, B. Zhou, D. Yue, X. Wang, Person search with natural language description, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2017, pp. 1970\u20131979.","DOI":"10.1109\/CVPR.2017.551"},{"key":"10.1016\/j.imavis.2026.105999_b27","series-title":"Semantically self-aligned network for text-to-image part-aware person re-identification","author":"Ding","year":"2021"},{"key":"10.1016\/j.imavis.2026.105999_b28","doi-asserted-by":"crossref","unstructured":"A. Zhu, Z. Wang, Y. Li, X. Wan, J. Jin, T. Wang, F. Hu, G. Hua, Dssl: Deep surroundings-person separation learning for text-based person retrieval, in: Proceedings of the 29th ACM International Conference on Multimedia, 2021, pp. 209\u2013217.","DOI":"10.1145\/3474085.3475369"},{"key":"10.1016\/j.imavis.2026.105999_b29","doi-asserted-by":"crossref","unstructured":"Z. Wang, A. Zhu, J. Xue, X. Wan, C. Liu, T. Wang, Y. Li, Look before you leap: Improving text-based person retrieval by learning a consistent cross-modal common manifold, in: Proceedings of the 30th ACM International Conference on Multimedia, 2022, pp. 1984\u20131992.","DOI":"10.1145\/3503161.3548166"},{"key":"10.1016\/j.imavis.2026.105999_b30","doi-asserted-by":"crossref","unstructured":"Z. Wang, A. Zhu, J. Xue, X. Wan, C. Liu, T. Wang, Y. Li, Caibc: Capturing all-round information beyond color for text-based person retrieval, in: Proceedings of the 30th ACM International Conference on Multimedia, 2022, pp. 5314\u20135322.","DOI":"10.1145\/3503161.3548057"},{"key":"10.1016\/j.imavis.2026.105999_b31","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"2895","article-title":"Mgrl: Mutual-guidance representation learning for text-to-image person retrieval","author":"Lv","year":"2024"},{"key":"10.1016\/j.imavis.2026.105999_b32","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.124071","article-title":"Full-view salient feature mining and alignment for text-based person search","volume":"251","author":"Xie","year":"2024","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.imavis.2026.105999_b33","first-page":"14052","article-title":"Causality-inspired invariant representation learning for text-based person retrieval","volume":"vol. 38","author":"Liu","year":"2024"},{"key":"10.1016\/j.imavis.2026.105999_b34","doi-asserted-by":"crossref","unstructured":"S. Zhao, C. Gao, Y. Shao, W.-S. Zheng, N. Sang, Weakly supervised text-based person re-identification, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 11395\u201311404.","DOI":"10.1109\/ICCV48922.2021.01120"},{"key":"10.1016\/j.imavis.2026.105999_b35","doi-asserted-by":"crossref","first-page":"6032","DOI":"10.1109\/TIP.2023.3327924","article-title":"Clip-driven fine-grained text-image person re-identification","volume":"32","author":"Yan","year":"2023","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.imavis.2026.105999_b36","doi-asserted-by":"crossref","first-page":"163","DOI":"10.1109\/TIP.2023.3337653","article-title":"VGSG: Vision-guided semantic-group network for text-based person search","volume":"33","author":"He","year":"2023","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.imavis.2026.105999_b37","series-title":"European Conference on Computer Vision","first-page":"624","article-title":"See finer, see more: Implicit modality alignment for text-based person retrieval","author":"Shu","year":"2022"}],"container-title":["Image and Vision Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S026288562600106X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S026288562600106X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T18:23:24Z","timestamp":1783103004000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S026288562600106X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":37,"alternative-id":["S026288562600106X"],"URL":"https:\/\/doi.org\/10.1016\/j.imavis.2026.105999","relation":{},"ISSN":["0262-8856"],"issn-type":[{"value":"0262-8856","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Unsupervised cross-modal person search via text style normalization","name":"articletitle","label":"Article Title"},{"value":"Image and Vision Computing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.imavis.2026.105999","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"105999"}}