{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T23:08:52Z","timestamp":1778800132831,"version":"3.51.4"},"reference-count":50,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["93K172025K21"],"award-info":[{"award-number":["93K172025K21"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62206006"],"award-info":[{"award-number":["62206006"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003995","name":"Natural Science Foundation of Anhui Province","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100003995","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100009091","name":"Justus Liebig Universit\u00e4t Gie\u00dfen","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100009091","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100009558","name":"University Natural Science Research Project of Anhui Province","doi-asserted-by":"publisher","award":["2024AH051783"],"award-info":[{"award-number":["2024AH051783"]}],"id":[{"id":"10.13039\/501100009558","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Vision and Image Understanding"],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1016\/j.cviu.2026.104733","type":"journal-article","created":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T16:14:45Z","timestamp":1773159285000},"page":"104733","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Text-to-image Person Search based on Semantic Reorganization"],"prefix":"10.1016","volume":"267","author":[{"given":"Jielong","family":"He","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feng","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiwen","family":"Qu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.cviu.2026.104733_b1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/j.ins.2020.11.004","article-title":"A negative transfer approach to person re-identification via domain augmentation","volume":"549","author":"Chen","year":"2021","journal-title":"Inform. Sci."},{"key":"10.1016\/j.cviu.2026.104733_b2","doi-asserted-by":"crossref","first-page":"48","DOI":"10.1016\/j.neucom.2020.07.087","article-title":"Self-supervised data augmentation for person re-identification","volume":"415","author":"Chen","year":"2020","journal-title":"Neurocomputing"},{"key":"10.1016\/j.cviu.2026.104733_b3","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109369","article-title":"Unsupervised person re-identification via multi-domain joint learning","volume":"138","author":"Chen","year":"2023","journal-title":"Pattern Recognit."},{"issue":"11","key":"10.1016\/j.cviu.2026.104733_b4","doi-asserted-by":"crossref","first-page":"12994","DOI":"10.1109\/TII.2024.3431044","article-title":"Ef-detr: A lightweight transformer-based object detector with an encoder-free neck","volume":"20","author":"Cheng","year":"2024","journal-title":"IEEE Trans. Ind. Informatics"},{"key":"10.1016\/j.cviu.2026.104733_b5","series-title":"Semantically self-aligned network for text-to-image part-aware person re-identification","author":"Ding","year":"2021"},{"key":"10.1016\/j.cviu.2026.104733_b6","doi-asserted-by":"crossref","unstructured":"Duan,\u00a0Y., Gu,\u00a0Z., Ying,\u00a0Z., Qi,\u00a0L., Meng,\u00a0C., Shi,\u00a0Y., 2024. Pc2: Pseudo-classification based pseudo-captioning for noisy correspondence learning in cross-modal retrieval. In: Proceedings of the 32nd ACM International Conference on Multimedia. pp. 9397\u20139406.","DOI":"10.1145\/3664647.3680860"},{"key":"10.1016\/j.cviu.2026.104733_b7","first-page":"4477","article-title":"Axm-net: Implicit cross-modal feature alignment for person re-identification","volume":"vol. 36","author":"Farooq","year":"2022"},{"key":"10.1016\/j.cviu.2026.104733_b8","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111510","article-title":"Learning multi-granularity representation with transformer for visible-infrared person re-identification","volume":"164","author":"Feng","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.cviu.2026.104733_b9","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110981","article-title":"Homogeneous and heterogeneous relational graph for visible-infrared person re-identification","volume":"158","author":"Feng","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.cviu.2026.104733_b10","doi-asserted-by":"crossref","unstructured":"Fu,\u00a0D., Chen,\u00a0D., Bao,\u00a0J., Yang,\u00a0H., Yuan,\u00a0L., Zhang,\u00a0L., Li,\u00a0H., Chen,\u00a0D., 2021. Unsupervised pre-training for person re-identification. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 14750\u201314759.","DOI":"10.1109\/CVPR46437.2021.01451"},{"key":"10.1016\/j.cviu.2026.104733_b11","unstructured":"Gong,\u00a0T., Wang,\u00a0J., Zhang,\u00a0L., 2024. Enhancing Cross-modal Completion and Alignment for Unsupervised Incomplete Text-to-Image Person Retrieval. In: Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence. pp. 794\u2013802."},{"key":"10.1016\/j.cviu.2026.104733_b12","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2023.103703","article-title":"Spectrum-irrelevant fine-grained representation for visible\u2013infrared person re-identification","volume":"232","author":"Gong","year":"2023","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.cviu.2026.104733_b13","doi-asserted-by":"crossref","first-page":"163","DOI":"10.1109\/TIP.2023.3337653","article-title":"VGSG: Vision-guided semantic-group network for text-based person search","volume":"33","author":"He","year":"2023","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.cviu.2026.104733_b14","doi-asserted-by":"crossref","unstructured":"Jiang,\u00a0D., Ye,\u00a0M., 2023. Cross-Modal Implicit Relation Reasoning and Aligning for Text-to-Image Person Retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 2787\u20132797.","DOI":"10.1109\/CVPR52729.2023.00273"},{"key":"10.1016\/j.cviu.2026.104733_b15","first-page":"3172","article-title":"Adaptive uncertainty-based learning for text-based person retrieval","volume":"vol. 38","author":"Li","year":"2024"},{"key":"10.1016\/j.cviu.2026.104733_b16","series-title":"Data augmentation for text-based person retrieval using large language models","author":"Li","year":"2024"},{"key":"10.1016\/j.cviu.2026.104733_b17","doi-asserted-by":"crossref","unstructured":"Li,\u00a0S., Xiao,\u00a0T., Li,\u00a0H., Zhou,\u00a0B., Yue,\u00a0D., Wang,\u00a0X., 2017. Person search with natural language description. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 1970\u20131979.","DOI":"10.1109\/CVPR.2017.551"},{"key":"10.1016\/j.cviu.2026.104733_b18","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2024.104192","article-title":"A simple but effective vision transformer framework for visible\u2013infrared person re-identification","volume":"249","author":"Li","year":"2024","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.cviu.2026.104733_b19","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109636","article-title":"BDNet: A BERT-based dual-path network for text-to-image cross-modal person re-identification","volume":"141","author":"Liu","year":"2023","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.cviu.2026.104733_b20","first-page":"5703","article-title":"DM-Adapter: Domain-aware mixture-of-adapters for text-based person retrieval","volume":"vol. 39","author":"Liu","year":"2025"},{"key":"10.1016\/j.cviu.2026.104733_b21","first-page":"14052","article-title":"Causality-inspired invariant representation learning for text-based person retrieval","volume":"vol. 38","author":"Liu","year":"2024"},{"key":"10.1016\/j.cviu.2026.104733_b22","doi-asserted-by":"crossref","first-page":"6409","DOI":"10.1109\/TIFS.2024.3417251","article-title":"Mind the inconsistent semantics in positive pairs: Semantic aligning and multimodal contrastive learning for text-based pedestrian search","volume":"19","author":"Lu","year":"2024","journal-title":"IEEE Trans. Inf. Forensics Secur."},{"key":"10.1016\/j.cviu.2026.104733_b23","doi-asserted-by":"crossref","first-page":"7181","DOI":"10.1109\/TIFS.2025.3586488","article-title":"Prompt-guided transformer and MLLM interactive learning for text-based pedestrian search","volume":"20","author":"Lu","year":"2025","journal-title":"IEEE Trans. Inf. Forensics Secur."},{"key":"10.1016\/j.cviu.2026.104733_b24","series-title":"LLaVA-ReID: Selective multi-image questioner for interactive person re-identification","author":"Lu","year":"2025"},{"key":"10.1016\/j.cviu.2026.104733_b25","doi-asserted-by":"crossref","unstructured":"L\u00fclf,\u00a0C., Lima\u00a0Martins,\u00a0D.M., Vaz\u00a0Salles,\u00a0M.A., Zhou,\u00a0Y., Gieseke,\u00a0F., 2024. CLIP-Branches: Interactive Fine-Tuning for Text-Image Retrieval. In: Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval. pp. 2719\u20132723.","DOI":"10.1145\/3626772.3657678"},{"key":"10.1016\/j.cviu.2026.104733_b26","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"2895","article-title":"MGRL: Mutual-guidance representation learning for text-to-image person retrieval","author":"Lv","year":"2024"},{"key":"10.1016\/j.cviu.2026.104733_b27","series-title":"European Conference on Computer Vision","first-page":"474","article-title":"PLOT: Text-based person search with part slot attention for corresponding part discovery","author":"Park","year":"2024"},{"key":"10.1016\/j.cviu.2026.104733_b28","doi-asserted-by":"crossref","unstructured":"Shao,\u00a0Z., Zhang,\u00a0X., Fang,\u00a0M., Lin,\u00a0Z., Wang,\u00a0J., Ding,\u00a0C., 2022. Learning granularity-unified representations for text-to-image person re-identification. In: Proceedings of the 30th Acm International Conference on Multimedia. pp. 5566\u20135574.","DOI":"10.1145\/3503161.3548028"},{"key":"10.1016\/j.cviu.2026.104733_b29","first-page":"4943","article-title":"Diverse person: Customize your own dataset for text-based person search","volume":"vol. 38","author":"Song","year":"2024"},{"key":"10.1016\/j.cviu.2026.104733_b30","first-page":"1","article-title":"Boundary-aware feature fusion with dual-stream attention for remote sensing small object detection","volume":"63","author":"Song","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.cviu.2026.104733_b31","doi-asserted-by":"crossref","unstructured":"Tan,\u00a0W., Ding,\u00a0C., Jiang,\u00a0J., Wang,\u00a0F., Zhan,\u00a0Y., Tao,\u00a0D., 2024. Harnessing the power of mllms for transferable text-to-image person reid. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 17127\u201317137.","DOI":"10.1109\/CVPR52733.2024.01621"},{"key":"10.1016\/j.cviu.2026.104733_b32","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0R., Chen,\u00a0F., Tang,\u00a0J., Yan,\u00a0P., 2022. Adaptive camera margin for mask-guided domain adaptive person re-identification. In: Proceedings of the 30th ACM International Conference on Multimedia. pp. 668\u2013677.","DOI":"10.1145\/3503161.3548216"},{"key":"10.1016\/j.cviu.2026.104733_b33","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0C., Luo,\u00a0Z., Lin,\u00a0Y., Li,\u00a0S., 2021. Text-based Person Search via Multi-Granularity Embedding Learning. In: IJCAI. pp. 1068\u20131074.","DOI":"10.24963\/ijcai.2021\/148"},{"key":"10.1016\/j.cviu.2026.104733_b34","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0D., Yan,\u00a0F., Wang,\u00a0Y., Zhao,\u00a0L., Liang,\u00a0X., Zhong,\u00a0H., Zhang,\u00a0R., 2024. Fine-grained Semantics-aware Representation Learning for Text-based Person Retrieval. In: Proceedings of the 2024 International Conference on Multimedia Retrieval. pp. 92\u2013100.","DOI":"10.1145\/3652583.3658054"},{"issue":"6","key":"10.1016\/j.cviu.2026.104733_b35","doi-asserted-by":"crossref","first-page":"5118","DOI":"10.1109\/TCSVT.2023.3340102","article-title":"Multiple information embedded hashing for large-scale cross-modal retrieval","volume":"34","author":"Wang","year":"2023","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.cviu.2026.104733_b36","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0Z., Zhu,\u00a0A., Xue,\u00a0J., Wan,\u00a0X., Liu,\u00a0C., Wang,\u00a0T., Li,\u00a0Y., 2022. Look before you leap: Improving text-based person retrieval by learning a consistent cross-modal common manifold. In: Proceedings of the 30th ACM International Conference on Multimedia. pp. 1984\u20131992.","DOI":"10.1145\/3503161.3548166"},{"issue":"8","key":"10.1016\/j.cviu.2026.104733_b37","doi-asserted-by":"crossref","first-page":"7005","DOI":"10.1109\/TCSVT.2023.3329220","article-title":"Contrastive transformer learning with proximity data generation for text-based person search","volume":"34","author":"Wu","year":"2023","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.cviu.2026.104733_b38","first-page":"5223","article-title":"Cap4Video++: Enhancing video understanding with auxiliary captions","author":"Wu","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104733_b39","doi-asserted-by":"crossref","first-page":"6032","DOI":"10.1109\/TIP.2023.3327924","article-title":"Clip-driven fine-grained text-image person re-identification","volume":"32","author":"Yan","year":"2023","journal-title":"IEEE Trans. Image Process."},{"issue":"12","key":"10.1016\/j.cviu.2026.104733_b40","doi-asserted-by":"crossref","first-page":"17973","DOI":"10.1109\/TNNLS.2023.3310118","article-title":"Image-specific information suppression and implicit local alignment for text-based person search","volume":"35","author":"Yan","year":"2024","journal-title":"IEEE Trans. Neural Networks Learn. Syst."},{"key":"10.1016\/j.cviu.2026.104733_b41","doi-asserted-by":"crossref","unstructured":"Yang,\u00a0S., Zhou,\u00a0Y., Zheng,\u00a0Z., Wang,\u00a0Y., Zhu,\u00a0L., Wu,\u00a0Y., 2023. Towards unified text-based person retrieval: A large-scale multi-attribute and language search benchmark. In: Proceedings of the 31st ACM International Conference on Multimedia. pp. 4492\u20134501.","DOI":"10.1145\/3581783.3611709"},{"key":"10.1016\/j.cviu.2026.104733_b42","doi-asserted-by":"crossref","first-page":"5465","DOI":"10.1109\/TIFS.2025.3573185","article-title":"Diverse co-saliency feature learning for text-based person retrieval","volume":"20","author":"You","year":"2025","journal-title":"IEEE Trans. Inf. Forensics Secur."},{"key":"10.1016\/j.cviu.2026.104733_b43","doi-asserted-by":"crossref","first-page":"3115","DOI":"10.1109\/TIP.2024.3390984","article-title":"Multi-granularity contrastive cross-modal collaborative generation for end-to-end long-term video question answering","volume":"33","author":"Yu","year":"2024","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.cviu.2026.104733_b44","doi-asserted-by":"crossref","unstructured":"Zhang,\u00a0Y., Lu,\u00a0H., 2018. Deep cross-modal projection learning for image-text matching. In: Proceedings of the European Conference on Computer Vision. ECCV, pp. 686\u2013701.","DOI":"10.1007\/978-3-030-01246-5_42"},{"key":"10.1016\/j.cviu.2026.104733_b45","doi-asserted-by":"crossref","unstructured":"Zheng,\u00a0Z., Zheng,\u00a0L., Yang,\u00a0Y., 2017. Unlabeled samples generated by gan improve the person re-identification baseline in vitro. In: Proceedings of the IEEE International Conference on Computer Vision. pp. 3754\u20133762.","DOI":"10.1109\/ICCV.2017.405"},{"issue":"11","key":"10.1016\/j.cviu.2026.104733_b46","doi-asserted-by":"crossref","first-page":"8282","DOI":"10.1109\/TII.2025.3588622","article-title":"GAANet: Graph aggregation alignment feature fusion for multispectral object detection","volume":"21","author":"Zheng","year":"2025","journal-title":"IEEE Trans. Ind. Informatics"},{"key":"10.1016\/j.cviu.2026.104733_b47","doi-asserted-by":"crossref","unstructured":"Zhong,\u00a0Z., Zheng,\u00a0L., Cao,\u00a0D., Li,\u00a0S., 2017. Re-ranking person re-identification with k-reciprocal encoding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 1318\u20131327.","DOI":"10.1109\/CVPR.2017.389"},{"issue":"2","key":"10.1016\/j.cviu.2026.104733_b48","doi-asserted-by":"crossref","first-page":"369","DOI":"10.1109\/TBC.2022.3215249","article-title":"An end-to-end blind image quality assessment method using a recurrent network and self-attention","volume":"69","author":"Zhou","year":"2022","journal-title":"IEEE Trans. Broadcast."},{"key":"10.1016\/j.cviu.2026.104733_b49","doi-asserted-by":"crossref","first-page":"3242","DOI":"10.1007\/s11263-024-02338-7","article-title":"Blind image quality assessment: Exploring content fidelity perceptibility via quality adversarial learning","author":"Zhou","year":"2025","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.cviu.2026.104733_b50","doi-asserted-by":"crossref","unstructured":"Zhu,\u00a0A., Wang,\u00a0Z., Li,\u00a0Y., Wan,\u00a0X., Jin,\u00a0J., Wang,\u00a0T., Hu,\u00a0F., Hua,\u00a0G., 2021. Dssl: Deep surroundings-person separation learning for text-based person retrieval. In: Proceedings of the 29th ACM International Conference on Multimedia. pp. 209\u2013217.","DOI":"10.1145\/3474085.3475369"}],"container-title":["Computer Vision and Image Understanding"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001001?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001001?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T22:30:01Z","timestamp":1778797801000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1077314226001001"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4]]},"references-count":50,"alternative-id":["S1077314226001001"],"URL":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104733","relation":{},"ISSN":["1077-3142"],"issn-type":[{"value":"1077-3142","type":"print"}],"subject":[],"published":{"date-parts":[[2026,4]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Text-to-image Person Search based on Semantic Reorganization","name":"articletitle","label":"Article Title"},{"value":"Computer Vision and Image Understanding","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104733","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104733"}}